Commit 08a7e391b14 for nodejs
commit 08a7e391b14f0889e0a3b5c4a3f02633070acb8b
Author: Daniel Lemire <daniel@lemire.me>
Date: Thu Oct 8 20:01:38 2026 -0400
deps: update simdjson to 5.0.3
PR-URL: https://github.com/nodejs/node/pull/66620
Refs: https://github.com/simdjson/simdjson/pull/2913
Refs: https://github.com/nodejs/node/pull/66495
Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
Reviewed-By: Antoine du Hamel <duhamelantoine1995@gmail.com>
Reviewed-By: Richard Lau <richard.lau@ibm.com>
diff --git a/deps/simdjson/simdjson.cpp b/deps/simdjson/simdjson.cpp
index 71f443b3d2b..980bb83cefa 100644
--- a/deps/simdjson/simdjson.cpp
+++ b/deps/simdjson/simdjson.cpp
@@ -1,4 +1,4 @@
-/* auto-generated on 2026-09-04 16:04:31 -0400. version 4.6.11 Do not edit! */
+/* auto-generated on 2026-10-07 22:43:29 -0400. version 5.0.3 Do not edit! */
/* including simdjson.cpp: */
/* begin file simdjson.cpp */
#define SIMDJSON_SRC_SIMDJSON_CPP
@@ -41,7 +41,9 @@
#endif
// C++ 26
-#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202402L) // update when the standard is finalized
+// While C++26 is a working draft, compilers report 202400L in C++26 mode
+// (both GCC 16 and Clang 21 do). Update when the standard is finalized.
+#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202400L)
#define SIMDJSON_CPLUSPLUS26 1
#endif
@@ -98,14 +100,48 @@
#endif
#endif
-// The current specification is unclear on how we detect
-// static reflection, both __cpp_lib_reflection and
-// __cpp_impl_reflection are proposed in the draft specification.
-// For now, we disable static reflect by default. It must be
-// specified at compiler time.
+// Static reflection.
+//
+// The reflection-based APIs (simdjson::to, document::get<T>, the builder,
+// compile-time JSON, annotations) need considerably more than the reflection
+// operator. We turn them on only when the compiler advertises all of:
+//
+// P2996 reflection (^^, splicers, <meta>) __cpp_impl_reflection,
+// __cpp_lib_reflection
+// P1306 expansion statements (template for) __cpp_expansion_statements
+// P3491 std::define_static_string / _array __cpp_lib_define_static
+//
+// Two further features we rely on have, as of this writing, no feature-test
+// macro of their own, so they cannot be checked directly:
+//
+// P3394 annotations ([[=x]], std::meta::annotations_of) -- used for
+// the annotations of simdjson/annotations.h (rename, skip, ...).
+// P3289 consteval blocks (consteval { ... }) -- used by compile_time_json.
+//
+// Every implementation that defines the four macros above also implements
+// those two, so requiring the four is sufficient in practice. If that ever
+// stops being true, define SIMDJSON_STATIC_REFLECTION=0 to opt out.
+//
+// SIMDJSON_STATIC_REFLECTION may always be defined by the user (or by the
+// build system) to 0 or 1 to override the detection.
+//
+// Note that C++26 mode alone is not enough: GCC 16 requires -freflection,
+// and only then does it define __cpp_impl_reflection.
#ifndef SIMDJSON_STATIC_REFLECTION
-#define SIMDJSON_STATIC_REFLECTION 0 // disabled by default.
+#if defined(SIMDJSON_CPLUSPLUS26) && \
+ defined(__cpp_impl_reflection) && __cpp_impl_reflection >= 202506L && \
+ defined(__cpp_lib_reflection) && __cpp_lib_reflection >= 202506L && \
+ defined(__cpp_expansion_statements) && \
+ __cpp_expansion_statements >= 202506L && \
+ defined(__cpp_lib_define_static) && __cpp_lib_define_static >= 202506L
+// __cpp_lib_reflection is the feature-test macro for <meta>, so there is no
+// need for a separate __has_include check (which would have to be guarded for
+// compilers that lack __has_include).
+#define SIMDJSON_STATIC_REFLECTION 1
+#else
+#define SIMDJSON_STATIC_REFLECTION 0
#endif
+#endif // SIMDJSON_STATIC_REFLECTION
#if defined(__apple_build_version__)
#if __apple_build_version__ < 14000000
@@ -138,6 +174,47 @@
#define SIMDJSON_SUPPORTS_DESERIALIZATION 0
#endif
+// The C++20 char8_t type (and std::u8string/std::u8string_view) is available.
+// Because all strings that simdjson produces are valid UTF-8, we can offer
+// char8_t variants of our string accessors when this macro is set.
+#if !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+#if defined(__cpp_char8_t) && __cpp_char8_t >= 201811L
+#define SIMDJSON_SUPPORTS_CHAR8_T 1
+#else
+#define SIMDJSON_SUPPORTS_CHAR8_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+
+// The C++23 fixed-width floating-point types std::float32_t and std::float64_t
+// (<stdfloat>) are available. They are optional even in C++23: a compiler that
+// provides them predefines __STDCPP_FLOAT32_T__ and __STDCPP_FLOAT64_T__.
+// When these macros are set, we offer get_float32() and get_float64().
+#if !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+#if defined(__STDCPP_FLOAT32_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT32_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT32_T
+#define SIMDJSON_SUPPORTS_FLOAT32_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+
+#if !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+#if defined(__STDCPP_FLOAT64_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT64_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT64_T
+#define SIMDJSON_SUPPORTS_FLOAT64_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T || SIMDJSON_SUPPORTS_FLOAT64_T
+#include <stdfloat>
+#endif
+
#if !defined(SIMDJSON_CONSTEVAL)
#if defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
@@ -146,6 +223,18 @@
#define SIMDJSON_CONSTEVAL 0
#endif // defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
#endif // !defined(SIMDJSON_CONSTEVAL)
+
+// SIMDJSON_CONSTEXPR_STRING is 'constexpr' when the standard library supports
+// constexpr std::string (e.g., libstdc++ 12 or better), and empty otherwise. It
+// lets functions that build a std::string be constant expressions when possible
+// while still compiling against older standard libraries.
+#if !defined(SIMDJSON_CONSTEXPR_STRING)
+#if defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#define SIMDJSON_CONSTEXPR_STRING constexpr
+#else
+#define SIMDJSON_CONSTEXPR_STRING
+#endif // defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#endif // !defined(SIMDJSON_CONSTEXPR_STRING)
#endif // SIMDJSON_COMPILER_CHECK_H
/* end file simdjson/compiler_check.h */
/* including simdjson/portability.h: #include "simdjson/portability.h" */
@@ -437,16 +526,86 @@ using std::size_t;
#endif
#endif
+#ifndef SIMDJSON_HAS_UNISTD_H
+#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+#define SIMDJSON_HAS_UNISTD_H 1
+#else
+#define SIMDJSON_HAS_UNISTD_H 0
+#endif
+#endif
+
+// padded_memory_map availability.
+//
+// On POSIX platforms the class is always available: the implementation uses
+// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
+//
+// On Windows the class is disabled by default and must be explicitly
+// opted into by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. Enabling
+// it requires:
+// 1. `<windows.h>` has been included *before* `<simdjson.h>` (so that
+// this header can see the Win32 types and the `_WINDOWS_` include
+// guard),
+// 2. the compilation targets Windows 10, version 1803 or later
+// (i.e. `NTDDI_VERSION >= NTDDI_WIN10_RS4`, `0x0A000005`). This is
+// required because the implementation relies on the modern memory
+// APIs introduced with that version (`CreateFileMapping2` /
+// `MapViewOfFile3`),
+// 3. the link step pulls in an import library that exports those APIs,
+// typically `onecore.lib` (or `mincore.lib`).
+//
+// The `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS` CMake option arranges (1)-(3)
+// automatically when building simdjson with its own CMake. Consumers using
+// simdjson as a pre-built library are responsible for setting the macro,
+// the Windows version macros, and the link library themselves.
+//
+// If the opt-in conditions are not met on Windows, `padded_memory_map`
+// simply does not exist -- any attempt to use it fails at compile time
+// with an "unknown identifier" diagnostic rather than silently degrading.
+//
+// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
+// available in the current translation unit. Users may test this macro to
+// conditionally compile code that depends on padded_memory_map.
+#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
+ #if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+ #elif defined(_WINDOWS_) && defined(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS) && SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+ #else
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
+ #endif
+#endif
#endif // SIMDJSON_PORTABILITY_H
/* end file simdjson/portability.h */
+#include <cstddef>
namespace simdjson {
namespace internal {
+/**
+ * @private
+ * Scratch capacity that every caller of to_chars must provide.
+ *
+ * The emitted decimal is at most ~24 characters, but dragonbox() and
+ * format_buffer() intentionally write past the logical end with fixed-size
+ * 16/17-byte memcpy/memset operations so the compiler can inline them (no
+ * libc mem* dispatch with size-class branches). The extra bytes are required
+ * for safety of those over-writes; do not shrink this below 40.
+ * See src/to_chars.cpp and #2805.
+ */
+// Use an unscoped enum (not static constexpr / inline constexpr):
+// - C++11 targets (readme_examples11, quickstart11, ...) still include this header
+// - a static constexpr in the amalgamated simdjson.cpp TU is unused there
+// (only callers in headers use it) and trips -Wunused-const-variable -Werror
+enum : size_t { to_chars_buffer_size = 40 };
/**
* @private
* Our own implementation of the C++17 to_chars function.
* Defined in src/to_chars
+ *
+ * @note The buffer starting at first must have at least to_chars_buffer_size
+ * bytes of writable storage (see to_chars_buffer_size).
+ * @note The input number must be finite (NaN/Inf are not supported).
+ * @note The result is NOT null-terminated.
*/
char *to_chars(char *first, const char *last, double value);
/**
@@ -456,6 +615,12 @@ char *to_chars(char *first, const char *last, double value);
*/
double from_chars(const char *first) noexcept;
double from_chars(const char *first, const char* end) noexcept;
+/**
+ * @private
+ * Same as from_chars, but produces a correctly rounded binary32 (float) value.
+ * Defined in src/from_chars
+ */
+float from_chars_float(const char *first) noexcept;
}
#ifndef SIMDJSON_EXCEPTIONS
@@ -466,6 +631,10 @@ double from_chars(const char *first, const char* end) noexcept;
#endif
#endif
+#ifndef SIMDJSON_ENABLE_NAN_INF
+#define SIMDJSON_ENABLE_NAN_INF 0
+#endif
+
} // namespace simdjson
#if defined(__GNUC__)
@@ -481,16 +650,14 @@ double from_chars(const char *first, const char* end) noexcept;
// Align to N-byte boundary
#define SIMDJSON_ROUNDUP_N(a, n) (((a) + ((n)-1)) & ~((n)-1))
-#define SIMDJSON_ROUNDDOWN_N(a, n) ((a) & ~((n)-1))
-
-#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
#if SIMDJSON_REGULAR_VISUAL_STUDIO
// We could use [[deprecated]] but it requires C++14
#define simdjson_deprecated __declspec(deprecated)
#define simdjson_really_inline __forceinline
- #define simdjson_never_inline __declspec(noinline)
+ #define simdjson_never_inline inline __declspec(noinline)
+ #define simdjson_really_flatten [[msvc::flatten]]
#define simdjson_unused
#define simdjson_warn_unused
@@ -531,6 +698,7 @@ double from_chars(const char *first, const char* end) noexcept;
#define simdjson_really_inline inline __attribute__((always_inline))
#define simdjson_never_inline inline __attribute__((noinline))
+ #define simdjson_really_flatten [[gnu::flatten]]
#define simdjson_unused __attribute__((unused))
#define simdjson_warn_unused __attribute__((warn_unused_result))
@@ -607,6 +775,15 @@ double from_chars(const char *first, const char* end) noexcept;
#define simdjson_inline simdjson_really_inline
#endif
+#if defined(simdjson_flatten)
+ // Prefer the user's definition of simdjson_flatten; don't define it ourselves.
+#elif (defined(__GNUC__) && !defined(__OPTIMIZE__)) || (defined(_DEBUG) && _MSC_VER )
+ // Flattening can lead to significant code bloat and high compile times. Don't use it for unoptimized builds.
+ #define simdjson_flatten
+#else
+ #define simdjson_flatten simdjson_really_flatten
+#endif
+
#if SIMDJSON_VISUAL_STUDIO
/**
* Windows users need to do some extra work when building
@@ -2562,6 +2739,7 @@ enum error_code {
OUT_OF_BOUNDS, ///< Attempted to access location outside of document.
TRAILING_CONTENT, ///< Unexpected trailing content in the JSON input
OUT_OF_CAPACITY, ///< The capacity was exceeded, we cannot allocate enough memory.
+ UNKNOWN_FIELD, ///< JSON field does not map to any member of the target (see simdjson::deny_unknown_fields)
NUM_ERROR_CODES ///< Placeholder for end of error code list.
};
@@ -2924,6 +3102,7 @@ inline const std::string error_message(int error) noexcept;
#if SIMDJSON_SUPPORTS_CONCEPTS
#include <concepts>
+#include <string_view>
#include <type_traits>
namespace simdjson {
@@ -2965,6 +3144,19 @@ concept constructible_from_string_view = std::is_constructible_v<T, std::string_
&& !std::is_same_v<T, std::string_view>
&& std::is_default_constructible_v<T>;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * A C++20 char8_t string type such as std::u8string. Such types cannot be built
+ * from a std::string_view (the character types differ), so they need their own
+ * deserialization path, going through the u8 string accessors.
+ */
+template<typename T>
+concept constructible_from_u8string_view = std::is_constructible_v<T, std::u8string_view>
+ && !std::is_same_v<T, std::u8string_view>
+ && !std::is_constructible_v<T, std::string_view>
+ && std::is_default_constructible_v<T>;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename M>
concept string_view_keyed_map = string_view_like<typename M::key_type>
&& requires(std::remove_cvref_t<M>& m, typename M::key_type sv, typename M::mapped_type v) {
@@ -3059,9 +3251,15 @@ concept string_like =
// Concept that checks if a type is a container but not a string (because
// strings handling must be handled differently)
// Now uses iterator-based approach for broader container support
+//
+// Optional types are excluded on purpose. Since C++26 (P3168), std::optional
+// is itself a range, so without the exclusion an std::optional would match
+// both this concept and optional_type, making the container and the optional
+// overloads of atom()/append() ambiguous. See issue 2827.
template <typename T>
concept container_but_not_string =
- std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>;
+ std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::optional_type<T>;
@@ -3168,6 +3366,11 @@ struct fixed_string {
data[i] = str[i];
}
}
+ constexpr fixed_string(const unsigned char (&str)[N]) {
+ for (std::size_t i = 0; i < N; ++i) {
+ data[i] = static_cast<char>(str[i]);
+ }
+ }
char data[N];
constexpr std::string_view view() const { return {data, N - 1}; }
constexpr size_t size() const { return N ; }
@@ -3199,6 +3402,11 @@ struct string_constant {
#endif // SIMDJSON_CONSTEVALUTIL_H
/* end file simdjson/constevalutil.h */
+#if SIMDJSON_SUPPORTS_CHAR8_T
+#include <string>
+#include <string_view>
+#endif
+
/**
* @brief The top level simdjson namespace, containing everything the library provides.
*/
@@ -3208,6 +3416,10 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS
/** The maximum document size supported by simdjson. */
constexpr size_t SIMDJSON_MAXSIZE_BYTES = 0xFFFFFFFF;
+/** The maximum depth of nested objects and arrays supported by simdjson.
+ A depth of SIMDJSON_MAXSIZE_BYTES/2 is not reasonable and would be
+ adversarial, but it serves as an upper bound for validation purposes. */
+constexpr size_t SIMDJSON_MAX_DEPTH = SIMDJSON_MAXSIZE_BYTES/2;
/**
* The amount of padding needed in a buffer to parse JSON.
@@ -3233,6 +3445,30 @@ struct padded_string;
class padded_string_view;
enum class stage1_mode;
+/**
+ * Stream format for parse_many/iterate_many.
+ */
+enum class stream_format {
+ whitespace_delimited, ///< Whitespace-delimited JSON documents (default, includes NDJSON/JSONL)
+ json_sequence, ///< RFC 7464 JSON text sequences (RS-delimited)
+ comma_delimited, ///< Comma-separated JSON documents (e.g., `{...},{...},{...}`)
+ comma_delimited_array,///< A single JSON array whose elements are iterated as
+ ///< comma-separated documents (e.g., `[{...},{...},{...}]`).
+ ///< The parser strips the outer `[` / `]` plus any
+ ///< surrounding JSON whitespace (space, tab, LF, CR)
+ ///< and then behaves like `comma_delimited` over the
+ ///< remaining bytes.
+ newline_delimited ///< NDJSON/JSON Lines where each document occupies exactly
+ ///< one line: documents are separated by line feeds and no
+ ///< document contains a raw line feed. Same inputs as
+ ///< `whitespace_delimited`, but the stronger guarantee lets
+ ///< the parser find the end of a document without walking
+ ///< it. On ondemand `iterate_many`, an unread remainder may
+ ///< be skipped by jumping to the next line feed without
+ ///< structure-validating that remainder. Use
+ ///< `whitespace_delimited` if unsure.
+};
+
namespace internal {
template<typename T>
@@ -3243,6 +3479,52 @@ class tape_ref;
struct value128;
enum class tape_type;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Reinterpret a UTF-8 string as a C++20 std::u8string_view. No byte is copied
+ * or modified: char8_t and char have the same size, representation and
+ * alignment. Every string that simdjson produces is valid UTF-8, so this is a
+ * lossless view over the very same memory.
+ * @private
+ */
+simdjson_inline std::u8string_view as_u8string_view(std::string_view v) noexcept {
+ return std::u8string_view(reinterpret_cast<const char8_t *>(v.data()), v.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
+/**
+ * Assign a UTF-8 string to a string-like receiver. The general case simply
+ * assigns the std::string_view: it covers std::string and any user type that
+ * can be assigned from a std::string_view.
+ * @private
+ */
+template <typename string_type>
+simdjson_inline void assign_utf8(string_type &receiver, std::string_view content) noexcept {
+ receiver = content;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Assign a UTF-8 string to a char8_t-based string (e.g., std::u8string). This
+ * overload is more specialized than the general one, so overload resolution
+ * prefers it whenever the receiver holds char8_t.
+ * @private
+ */
+template <typename traits_type, typename allocator_type>
+simdjson_inline void assign_utf8(std::basic_string<char8_t, traits_type, allocator_type> &receiver, std::string_view content) noexcept {
+ receiver.assign(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+
+/**
+ * Assign a UTF-8 string to a char8_t-based string view (e.g., std::u8string_view).
+ * @private
+ */
+template <typename traits_type>
+simdjson_inline void assign_utf8(std::basic_string_view<char8_t, traits_type> &receiver, std::string_view content) noexcept {
+ receiver = std::basic_string_view<char8_t, traits_type>(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
} // namespace internal
} // namespace simdjson
@@ -3263,806 +3545,1038 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS
#include <cstring>
#include <cstdint>
-#include <array>
#include <cmath>
+#include <limits>
namespace simdjson {
namespace internal {
/*!
-implements the Grisu2 algorithm for binary to decimal floating-point
-conversion.
-Adapted from JSON for Modern C++
-
-This implementation is a slightly modified version of the reference
-implementation which may be obtained from
-http://florian.loitsch.com/publications (bench.tar.gz).
-The code is distributed under the MIT license, Copyright (c) 2009 Florian
-Loitsch. For a detailed description of the algorithm see: [1] Loitsch, "Printing
-Floating-Point Numbers Quickly and Accurately with Integers", Proceedings of the
-ACM SIGPLAN 2010 Conference on Programming Language Design and Implementation,
-PLDI 2010 [2] Burger, Dybvig, "Printing Floating-Point Numbers Quickly and
-Accurately", Proceedings of the ACM SIGPLAN 1996 Conference on Programming
-Language Design and Implementation, PLDI 1996
-*/
-namespace dtoa_impl {
-
-template <typename Target, typename Source>
-Target reinterpret_bits(const Source source) {
- static_assert(sizeof(Target) == sizeof(Source), "size mismatch");
-
- Target target;
- std::memcpy(&target, &source, sizeof(Source));
- return target;
-}
-
-struct diyfp // f * 2^e
-{
- static constexpr int kPrecision = 64; // = q
-
- std::uint64_t f = 0;
- int e = 0;
+Implements the Dragonbox algorithm for binary to decimal floating-point
+conversion (shortest round-trip representation of a double).
- constexpr diyfp(std::uint64_t f_, int e_) noexcept : f(f_), e(e_) {}
+The digit-generation core below is a self-contained port of Junekey Jeon's
+reference "simple_dragonbox" implementation, specialized to IEEE-754 binary64
+and de-templated to match simdjson's style. Only the shortest-representation
+path with the default (nearest, ties-to-even) rounding is kept.
- /*!
- @brief returns x - y
- @pre x.e == y.e and x.f >= y.f
- */
- static diyfp sub(const diyfp &x, const diyfp &y) noexcept {
+Dragonbox: https://github.com/jk-jeon/dragonbox
+Copyright 2020-2025 Junekey Jeon (and contributors).
- return {x.f - y.f, x.e};
- }
+The original is dual-licensed; this port is used under the terms of the
+Boost Software License, Version 1.0 (https://www.boost.org/LICENSE_1_0.txt).
- /*!
- @brief returns x * y
- @note The result is rounded. (Only the upper q bits are returned.)
- */
- static diyfp mul(const diyfp &x, const diyfp &y) noexcept {
- static_assert(kPrecision == 64, "internal error");
+For the algorithm itself see:
+[1] Junekey Jeon, "Dragonbox: A New Floating-Point Binary-to-Decimal Conversion Algorithm" (2022).
- // Computes:
- // f = round((x.f * y.f) / 2^q)
- // e = x.e + y.e + q
-
- // Emulate the 64-bit * 64-bit multiplication:
- //
- // p = u * v
- // = (u_lo + 2^32 u_hi) (v_lo + 2^32 v_hi)
- // = (u_lo v_lo ) + 2^32 ((u_lo v_hi ) + (u_hi v_lo )) +
- // 2^64 (u_hi v_hi ) = (p0 ) + 2^32 ((p1 ) + (p2 ))
- // + 2^64 (p3 ) = (p0_lo + 2^32 p0_hi) + 2^32 ((p1_lo +
- // 2^32 p1_hi) + (p2_lo + 2^32 p2_hi)) + 2^64 (p3 ) =
- // (p0_lo ) + 2^32 (p0_hi + p1_lo + p2_lo ) + 2^64 (p1_hi +
- // p2_hi + p3) = (p0_lo ) + 2^32 (Q ) + 2^64 (H ) = (p0_lo ) +
- // 2^32 (Q_lo + 2^32 Q_hi ) + 2^64 (H )
- //
- // (Since Q might be larger than 2^32 - 1)
- //
- // = (p0_lo + 2^32 Q_lo) + 2^64 (Q_hi + H)
- //
- // (Q_hi + H does not overflow a 64-bit int)
- //
- // = p_lo + 2^64 p_hi
-
- const std::uint64_t u_lo = x.f & 0xFFFFFFFFu;
- const std::uint64_t u_hi = x.f >> 32u;
- const std::uint64_t v_lo = y.f & 0xFFFFFFFFu;
- const std::uint64_t v_hi = y.f >> 32u;
-
- const std::uint64_t p0 = u_lo * v_lo;
- const std::uint64_t p1 = u_lo * v_hi;
- const std::uint64_t p2 = u_hi * v_lo;
- const std::uint64_t p3 = u_hi * v_hi;
-
- const std::uint64_t p0_hi = p0 >> 32u;
- const std::uint64_t p1_lo = p1 & 0xFFFFFFFFu;
- const std::uint64_t p1_hi = p1 >> 32u;
- const std::uint64_t p2_lo = p2 & 0xFFFFFFFFu;
- const std::uint64_t p2_hi = p2 >> 32u;
-
- std::uint64_t Q = p0_hi + p1_lo + p2_lo;
-
- // The full product might now be computed as
- //
- // p_hi = p3 + p2_hi + p1_hi + (Q >> 32)
- // p_lo = p0_lo + (Q << 32)
- //
- // But in this particular case here, the full p_lo is not required.
- // Effectively we only need to add the highest bit in p_lo to p_hi (and
- // Q_hi + 1 does not overflow).
-
- Q += std::uint64_t{1} << (64u - 32u - 1u); // round, ties up
-
- const std::uint64_t h = p3 + p2_hi + p1_hi + (Q >> 32u);
-
- return {h, x.e + y.e + 64};
- }
-
- /*!
- @brief normalize x such that the significand is >= 2^(q-1)
- @pre x.f != 0
- */
- static diyfp normalize(diyfp x) noexcept {
-
- while ((x.f >> 63u) == 0) {
- x.f <<= 1u;
- x.e--;
- }
-
- return x;
- }
-
- /*!
- @brief normalize x such that the result has the exponent E
- @pre e >= x.e and the upper e - x.e bits of x.f must be zero.
- */
- static diyfp normalize_to(const diyfp &x,
- const int target_exponent) noexcept {
- const int delta = x.e - target_exponent;
-
- return {x.f << delta, target_exponent};
- }
-};
-
-struct boundaries {
- diyfp w;
- diyfp minus;
- diyfp plus;
-};
-
-/*!
-Compute the (normalized) diyfp representing the input number 'value' and its
-boundaries.
-@pre value must be finite and positive
+The shortest decimal digits produced here are then laid out into the familiar
+printf("%g")-style text by format_buffer(), which is unchanged from the previous
+Grisu2-based implementation, so the emitted strings are identical except that
+Dragonbox always yields the (sometimes shorter) shortest representation.
*/
-template <typename FloatType> boundaries compute_boundaries(FloatType value) {
+namespace dtoa_impl {
- // Convert the IEEE representation into a diyfp.
- //
- // If v is denormal:
- // value = 0.F * 2^(1 - bias) = ( F) * 2^(1 - bias - (p-1))
- // If v is normalized:
- // value = 1.F * 2^(E - bias) = (2^(p-1) + F) * 2^(E - bias - (p-1))
+// 128-bit helpers (no compiler intrinsics, so the code stays portable).
+struct uint128 {
+ std::uint64_t high;
+ std::uint64_t low;
+};
- static_assert(std::numeric_limits<FloatType>::is_iec559,
- "internal error: dtoa_short requires an IEEE-754 "
- "floating-point implementation");
+inline std::uint64_t rotr64(std::uint64_t n, unsigned r) noexcept {
+ r &= 63;
+ return (n >> r) | (n << ((64 - r) & 63));
+}
- constexpr int kPrecision =
- std::numeric_limits<FloatType>::digits; // = p (includes the hidden bit)
- constexpr int kBias =
- std::numeric_limits<FloatType>::max_exponent - 1 + (kPrecision - 1);
- constexpr int kMinExp = 1 - kBias;
- constexpr std::uint64_t kHiddenBit = std::uint64_t{1}
- << (kPrecision - 1); // = 2^(p-1)
+inline std::uint64_t umul64(std::uint32_t x, std::uint32_t y) noexcept {
+ return x * std::uint64_t(y);
+}
- using bits_type = typename std::conditional<kPrecision == 24, std::uint32_t,
- std::uint64_t>::type;
+// 64x64 -> 128 bit multiplication.
+inline uint128 umul128(std::uint64_t x, std::uint64_t y) noexcept {
+#if defined(__SIZEOF_INT128__)
+ const __uint128_t p = static_cast<__uint128_t>(x)*y;
+ return {std::uint64_t(p>>64), std::uint64_t(p)};
+#else // using fallback on 32-bit targets and MSVC
+ const std::uint32_t a = std::uint32_t(x >> 32);
+ const std::uint32_t b = std::uint32_t(x);
+ const std::uint32_t c = std::uint32_t(y >> 32);
+ const std::uint32_t d = std::uint32_t(y);
- const std::uint64_t bits = reinterpret_bits<bits_type>(value);
- const std::uint64_t E = bits >> (kPrecision - 1);
- const std::uint64_t F = bits & (kHiddenBit - 1);
+ const std::uint64_t ac = umul64(a, c);
+ const std::uint64_t bc = umul64(b, c);
+ const std::uint64_t ad = umul64(a, d);
+ const std::uint64_t bd = umul64(b, d);
- const bool is_denormal = E == 0;
- const diyfp v = is_denormal
- ? diyfp(F, kMinExp)
- : diyfp(F + kHiddenBit, static_cast<int>(E) - kBias);
+ const std::uint64_t intermediate =
+ (bd >> 32) + std::uint32_t(ad) + std::uint32_t(bc);
- // Compute the boundaries m- and m+ of the floating-point value
- // v = f * 2^e.
- //
- // Determine v- and v+, the floating-point predecessor and successor if v,
- // respectively.
- //
- // v- = v - 2^e if f != 2^(p-1) or e == e_min (A)
- // = v - 2^(e-1) if f == 2^(p-1) and e > e_min (B)
- //
- // v+ = v + 2^e
- //
- // Let m- = (v- + v) / 2 and m+ = (v + v+) / 2. All real numbers _strictly_
- // between m- and m+ round to v, regardless of how the input rounding
- // algorithm breaks ties.
- //
- // ---+-------------+-------------+-------------+-------------+--- (A)
- // v- m- v m+ v+
- //
- // -----------------+------+------+-------------+-------------+--- (B)
- // v- m- v m+ v+
+ return {ac + (intermediate >> 32) + (ad >> 32) + (bc >> 32),
+ (intermediate << 32) + std::uint32_t(bd)};
+#endif
+}
- const bool lower_boundary_is_closer = F == 0 && E > 1;
- const diyfp m_plus = diyfp(2 * v.f + 1, v.e - 1);
- const diyfp m_minus = lower_boundary_is_closer
- ? diyfp(4 * v.f - 1, v.e - 2) // (B)
- : diyfp(2 * v.f - 1, v.e - 1); // (A)
+// High 64 bits of a 64x64 -> 128 bit multiplication.
+inline std::uint64_t umul128_upper64(std::uint64_t x, std::uint64_t y) noexcept {
+#if defined(__SIZEOF_INT128__)
+ return std::uint64_t((static_cast<__uint128_t>(x)*y)>>64);
+#else // using fallback on 32-bit targets and MSVC
+ const std::uint32_t a = std::uint32_t(x >> 32);
+ const std::uint32_t b = std::uint32_t(x);
+ const std::uint32_t c = std::uint32_t(y >> 32);
+ const std::uint32_t d = std::uint32_t(y);
- // Determine the normalized w+ = m+.
- const diyfp w_plus = diyfp::normalize(m_plus);
+ const std::uint64_t ac = umul64(a, c);
+ const std::uint64_t bc = umul64(b, c);
+ const std::uint64_t ad = umul64(a, d);
+ const std::uint64_t bd = umul64(b, d);
- // Determine w- = m- such that e_(w-) = e_(w+).
- const diyfp w_minus = diyfp::normalize_to(m_minus, w_plus.e);
+ const std::uint64_t intermediate =
+ (bd >> 32) + std::uint32_t(ad) + std::uint32_t(bc);
- return {diyfp::normalize(v), w_minus, w_plus};
+ return ac + (intermediate >> 32) + (ad >> 32) + (bc >> 32);
+#endif
}
-// Given normalized diyfp w, Grisu needs to find a (normalized) cached
-// power-of-ten c, such that the exponent of the product c * w = f * 2^e lies
-// within a certain range [alpha, gamma] (Definition 3.2 from [1])
-//
-// alpha <= e = e_c + e_w + q <= gamma
-//
-// or
-//
-// f_c * f_w * 2^alpha <= f_c 2^(e_c) * f_w 2^(e_w) * 2^q
-// <= f_c * f_w * 2^gamma
-//
-// Since c and w are normalized, i.e. 2^(q-1) <= f < 2^q, this implies
-//
-// 2^(q-1) * 2^(q-1) * 2^alpha <= c * w * 2^q < 2^q * 2^q * 2^gamma
-//
-// or
-//
-// 2^(q - 2 + alpha) <= c * w < 2^(q + gamma)
-//
-// The choice of (alpha,gamma) determines the size of the table and the form of
-// the digit generation procedure. Using (alpha,gamma)=(-60,-32) works out well
-// in practice:
-//
-// The idea is to cut the number c * w = f * 2^e into two parts, which can be
-// processed independently: An integral part p1, and a fractional part p2:
-//
-// f * 2^e = ( (f div 2^-e) * 2^-e + (f mod 2^-e) ) * 2^e
-// = (f div 2^-e) + (f mod 2^-e) * 2^e
-// = p1 + p2 * 2^e
-//
-// The conversion of p1 into decimal form requires a series of divisions and
-// modulos by (a power of) 10. These operations are faster for 32-bit than for
-// 64-bit integers, so p1 should ideally fit into a 32-bit integer. This can be
-// achieved by choosing
-//
-// -e >= 32 or e <= -32 := gamma
-//
-// In order to convert the fractional part
-//
-// p2 * 2^e = p2 / 2^-e = d[-1] / 10^1 + d[-2] / 10^2 + ...
-//
-// into decimal form, the fraction is repeatedly multiplied by 10 and the digits
-// d[-i] are extracted in order:
-//
-// (10 * p2) div 2^-e = d[-1]
-// (10 * p2) mod 2^-e = d[-2] / 10^1 + ...
-//
-// The multiplication by 10 must not overflow. It is sufficient to choose
-//
-// 10 * p2 < 16 * p2 = 2^4 * p2 <= 2^64.
-//
-// Since p2 = f mod 2^-e < 2^-e,
-//
-// -e <= 60 or e >= -60 := alpha
-
-constexpr int kAlpha = -60;
-constexpr int kGamma = -32;
+// Upper 128 bits of a 64 x 128 -> 192 bit multiplication.
+inline uint128 umul192_upper128(std::uint64_t x, uint128 y) noexcept {
+ uint128 r = umul128(x, y.high);
+ const std::uint64_t add = umul128_upper64(x, y.low);
+ const std::uint64_t sum = r.low + add;
+ r.high += (sum < r.low) ? 1 : 0;
+ r.low = sum;
+ return r;
+}
+
+// Lower 128 bits of a 64 x 128 -> 192 bit multiplication.
+inline uint128 umul192_lower128(std::uint64_t x, uint128 y) noexcept {
+ const std::uint64_t high = x * y.high;
+ const uint128 high_low = umul128(x, y.low);
+ return {high + high_low.high, high_low.low};
+}
+
+// Integer log approximations (exact over the range of inputs we feed them).
+inline int floor_log10_pow2(int e) noexcept { return (e * 315653) >> 20; }
+inline int floor_log2_pow10(int e) noexcept { return (e * 1741647) >> 19; }
+inline int floor_log10_pow2_minus_log10_4_over_3(int e) noexcept {
+ return (e * 631305 - 261663) >> 21;
+}
+
+// Format constants for IEEE-754 binary64, plus the precomputed cache of powers of ten.
+static constexpr int kappa = 2;
+static constexpr int significand_bits = 52;
+static constexpr int total_bits = 64;
+static constexpr int min_exponent = -1022;
+static constexpr int exponent_bias = -1023;
+static constexpr int cache_min_k = -292;
+static constexpr int big_divisor = 1000; // 10^(kappa + 1)
+static constexpr int small_divisor = 100; // 10^kappa
+static constexpr int case_shorter_interval_left_endpoint_lower_threshold = 2;
+static constexpr int case_shorter_interval_left_endpoint_upper_threshold = 3;
+static constexpr int shorter_interval_tie_lower_threshold = -77;
+static constexpr int shorter_interval_tie_upper_threshold = -77;
+
+// cache[i] holds a 128-bit approximation of a power of ten; indexed by
+// (-minus_k - cache_min_k). Taken verbatim from the Dragonbox reference.
+static constexpr uint128 cache[619] = {
+ {0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7b},
+ {0x9faacf3df73609b1, 0x77b191618c54e9ad},
+ {0xc795830d75038c1d, 0xd59df5b9ef6a2418},
+ {0xf97ae3d0d2446f25, 0x4b0573286b44ad1e},
+ {0x9becce62836ac577, 0x4ee367f9430aec33},
+ {0xc2e801fb244576d5, 0x229c41f793cda740},
+ {0xf3a20279ed56d48a, 0x6b43527578c11110},
+ {0x9845418c345644d6, 0x830a13896b78aaaa},
+ {0xbe5691ef416bd60c, 0x23cc986bc656d554},
+ {0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa9},
+ {0x94b3a202eb1c3f39, 0x7bf7d71432f3d6aa},
+ {0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc54},
+ {0xe858ad248f5c22c9, 0xd1b3400f8f9cff69},
+ {0x91376c36d99995be, 0x23100809b9c21fa2},
+ {0xb58547448ffffb2d, 0xabd40a0c2832a78b},
+ {0xe2e69915b3fff9f9, 0x16c90c8f323f516d},
+ {0x8dd01fad907ffc3b, 0xae3da7d97f6792e4},
+ {0xb1442798f49ffb4a, 0x99cd11cfdf41779d},
+ {0xdd95317f31c7fa1d, 0x40405643d711d584},
+ {0x8a7d3eef7f1cfc52, 0x482835ea666b2573},
+ {0xad1c8eab5ee43b66, 0xda3243650005eed0},
+ {0xd863b256369d4a40, 0x90bed43e40076a83},
+ {0x873e4f75e2224e68, 0x5a7744a6e804a292},
+ {0xa90de3535aaae202, 0x711515d0a205cb37},
+ {0xd3515c2831559a83, 0x0d5a5b44ca873e04},
+ {0x8412d9991ed58091, 0xe858790afe9486c3},
+ {0xa5178fff668ae0b6, 0x626e974dbe39a873},
+ {0xce5d73ff402d98e3, 0xfb0a3d212dc81290},
+ {0x80fa687f881c7f8e, 0x7ce66634bc9d0b9a},
+ {0xa139029f6a239f72, 0x1c1fffc1ebc44e81},
+ {0xc987434744ac874e, 0xa327ffb266b56221},
+ {0xfbe9141915d7a922, 0x4bf1ff9f0062baa9},
+ {0x9d71ac8fada6c9b5, 0x6f773fc3603db4aa},
+ {0xc4ce17b399107c22, 0xcb550fb4384d21d4},
+ {0xf6019da07f549b2b, 0x7e2a53a146606a49},
+ {0x99c102844f94e0fb, 0x2eda7444cbfc426e},
+ {0xc0314325637a1939, 0xfa911155fefb5309},
+ {0xf03d93eebc589f88, 0x793555ab7eba27cb},
+ {0x96267c7535b763b5, 0x4bc1558b2f3458df},
+ {0xbbb01b9283253ca2, 0x9eb1aaedfb016f17},
+ {0xea9c227723ee8bcb, 0x465e15a979c1cadd},
+ {0x92a1958a7675175f, 0x0bfacd89ec191eca},
+ {0xb749faed14125d36, 0xcef980ec671f667c},
+ {0xe51c79a85916f484, 0x82b7e12780e7401b},
+ {0x8f31cc0937ae58d2, 0xd1b2ecb8b0908811},
+ {0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa16},
+ {0xdfbdcece67006ac9, 0x67a791e093e1d49b},
+ {0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e1},
+ {0xaecc49914078536d, 0x58fae9f773886e19},
+ {0xda7f5bf590966848, 0xaf39a475506a899f},
+ {0x888f99797a5e012d, 0x6d8406c952429604},
+ {0xaab37fd7d8f58178, 0xc8e5087ba6d33b84},
+ {0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a65},
+ {0x855c3be0a17fcd26, 0x5cf2eea09a550680},
+ {0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481f},
+ {0xd0601d8efc57b08b, 0xf13b94daf124da27},
+ {0x823c12795db6ce57, 0x76c53d08d6b70859},
+ {0xa2cb1717b52481ed, 0x54768c4b0c64ca6f},
+ {0xcb7ddcdda26da268, 0xa9942f5dcf7dfd0a},
+ {0xfe5d54150b090b02, 0xd3f93b35435d7c4d},
+ {0x9efa548d26e5a6e1, 0xc47bc5014a1a6db0},
+ {0xc6b8e9b0709f109a, 0x359ab6419ca1091c},
+ {0xf867241c8cc6d4c0, 0xc30163d203c94b63},
+ {0x9b407691d7fc44f8, 0x79e0de63425dcf1e},
+ {0xc21094364dfb5636, 0x985915fc12f542e5},
+ {0xf294b943e17a2bc4, 0x3e6f5b7b17b2939e},
+ {0x979cf3ca6cec5b5a, 0xa705992ceecf9c43},
+ {0xbd8430bd08277231, 0x50c6ff782a838354},
+ {0xece53cec4a314ebd, 0xa4f8bf5635246429},
+ {0x940f4613ae5ed136, 0x871b7795e136be9a},
+ {0xb913179899f68584, 0x28e2557b59846e40},
+ {0xe757dd7ec07426e5, 0x331aeada2fe589d0},
+ {0x9096ea6f3848984f, 0x3ff0d2c85def7622},
+ {0xb4bca50b065abe63, 0x0fed077a756b53aa},
+ {0xe1ebce4dc7f16dfb, 0xd3e8495912c62895},
+ {0x8d3360f09cf6e4bd, 0x64712dd7abbbd95d},
+ {0xb080392cc4349dec, 0xbd8d794d96aacfb4},
+ {0xdca04777f541c567, 0xecf0d7a0fc5583a1},
+ {0x89e42caaf9491b60, 0xf41686c49db57245},
+ {0xac5d37d5b79b6239, 0x311c2875c522ced6},
+ {0xd77485cb25823ac7, 0x7d633293366b828c},
+ {0x86a8d39ef77164bc, 0xae5dff9c02033198},
+ {0xa8530886b54dbdeb, 0xd9f57f830283fdfd},
+ {0xd267caa862a12d66, 0xd072df63c324fd7c},
+ {0x8380dea93da4bc60, 0x4247cb9e59f71e6e},
+ {0xa46116538d0deb78, 0x52d9be85f074e609},
+ {0xcd795be870516656, 0x67902e276c921f8c},
+ {0x806bd9714632dff6, 0x00ba1cd8a3db53b7},
+ {0xa086cfcd97bf97f3, 0x80e8a40eccd228a5},
+ {0xc8a883c0fdaf7df0, 0x6122cd128006b2ce},
+ {0xfad2a4b13d1b5d6c, 0x796b805720085f82},
+ {0x9cc3a6eec6311a63, 0xcbe3303674053bb1},
+ {0xc3f490aa77bd60fc, 0xbedbfc4411068a9d},
+ {0xf4f1b4d515acb93b, 0xee92fb5515482d45},
+ {0x991711052d8bf3c5, 0x751bdd152d4d1c4b},
+ {0xbf5cd54678eef0b6, 0xd262d45a78a0635e},
+ {0xef340a98172aace4, 0x86fb897116c87c35},
+ {0x9580869f0e7aac0e, 0xd45d35e6ae3d4da1},
+ {0xbae0a846d2195712, 0x8974836059cca10a},
+ {0xe998d258869facd7, 0x2bd1a438703fc94c},
+ {0x91ff83775423cc06, 0x7b6306a34627ddd0},
+ {0xb67f6455292cbf08, 0x1a3bc84c17b1d543},
+ {0xe41f3d6a7377eeca, 0x20caba5f1d9e4a94},
+ {0x8e938662882af53e, 0x547eb47b7282ee9d},
+ {0xb23867fb2a35b28d, 0xe99e619a4f23aa44},
+ {0xdec681f9f4c31f31, 0x6405fa00e2ec94d5},
+ {0x8b3c113c38f9f37e, 0xde83bc408dd3dd05},
+ {0xae0b158b4738705e, 0x9624ab50b148d446},
+ {0xd98ddaee19068c76, 0x3badd624dd9b0958},
+ {0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d7},
+ {0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4d},
+ {0xd47487cc8470652b, 0x7647c32000696720},
+ {0x84c8d4dfd2c63f3b, 0x29ecd9f40041e074},
+ {0xa5fb0a17c777cf09, 0xf468107100525891},
+ {0xcf79cc9db955c2cc, 0x7182148d4066eeb5},
+ {0x81ac1fe293d599bf, 0xc6f14cd848405531},
+ {0xa21727db38cb002f, 0xb8ada00e5a506a7d},
+ {0xca9cf1d206fdc03b, 0xa6d90811f0e4851d},
+ {0xfd442e4688bd304a, 0x908f4a166d1da664},
+ {0x9e4a9cec15763e2e, 0x9a598e4e043287ff},
+ {0xc5dd44271ad3cdba, 0x40eff1e1853f29fe},
+ {0xf7549530e188c128, 0xd12bee59e68ef47d},
+ {0x9a94dd3e8cf578b9, 0x82bb74f8301958cf},
+ {0xc13a148e3032d6e7, 0xe36a52363c1faf02},
+ {0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac2},
+ {0x96f5600f15a7b7e5, 0x29ab103a5ef8c0ba},
+ {0xbcb2b812db11a5de, 0x7415d448f6b6f0e8},
+ {0xebdf661791d60f56, 0x111b495b3464ad22},
+ {0x936b9fcebb25c995, 0xcab10dd900beec35},
+ {0xb84687c269ef3bfb, 0x3d5d514f40eea743},
+ {0xe65829b3046b0afa, 0x0cb4a5a3112a5113},
+ {0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ac},
+ {0xb3f4e093db73a093, 0x59ed216765690f57},
+ {0xe0f218b8d25088b8, 0x306869c13ec3532d},
+ {0x8c974f7383725573, 0x1e414218c73a13fc},
+ {0xafbd2350644eeacf, 0xe5d1929ef90898fb},
+ {0xdbac6c247d62a583, 0xdf45f746b74abf3a},
+ {0x894bc396ce5da772, 0x6b8bba8c328eb784},
+ {0xab9eb47c81f5114f, 0x066ea92f3f326565},
+ {0xd686619ba27255a2, 0xc80a537b0efefebe},
+ {0x8613fd0145877585, 0xbd06742ce95f5f37},
+ {0xa798fc4196e952e7, 0x2c48113823b73705},
+ {0xd17f3b51fca3a7a0, 0xf75a15862ca504c6},
+ {0x82ef85133de648c4, 0x9a984d73dbe722fc},
+ {0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebbb},
+ {0xcc963fee10b7d1b3, 0x318df905079926a9},
+ {0xffbbcfe994e5c61f, 0xfdf17746497f7053},
+ {0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa634},
+ {0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc1},
+ {0xf9bd690a1b68637b, 0x3dfdce7aa3c673b1},
+ {0x9c1661a651213e2d, 0x06bea10ca65c084f},
+ {0xc31bfa0fe5698db8, 0x486e494fcff30a63},
+ {0xf3e2f893dec3f126, 0x5a89dba3c3efccfb},
+ {0x986ddb5c6b3a76b7, 0xf89629465a75e01d},
+ {0xbe89523386091465, 0xf6bbb397f1135824},
+ {0xee2ba6c0678b597f, 0x746aa07ded582e2d},
+ {0x94db483840b717ef, 0xa8c2a44eb4571cdd},
+ {0xba121a4650e4ddeb, 0x92f34d62616ce414},
+ {0xe896a0d7e51e1566, 0x77b020baf9c81d18},
+ {0x915e2486ef32cd60, 0x0ace1474dc1d122f},
+ {0xb5b5ada8aaff80b8, 0x0d819992132456bb},
+ {0xe3231912d5bf60e6, 0x10e1fff697ed6c6a},
+ {0x8df5efabc5979c8f, 0xca8d3ffa1ef463c2},
+ {0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb3},
+ {0xddd0467c64bce4a0, 0xac7cb3f6d05ddbdf},
+ {0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96c},
+ {0xad4ab7112eb3929d, 0x86c16c98d2c953c7},
+ {0xd89d64d57a607744, 0xe871c7bf077ba8b8},
+ {0x87625f056c7c4a8b, 0x11471cd764ad4973},
+ {0xa93af6c6c79b5d2d, 0xd598e40d3dd89bd0},
+ {0xd389b47879823479, 0x4aff1d108d4ec2c4},
+ {0x843610cb4bf160cb, 0xcedf722a585139bb},
+ {0xa54394fe1eedb8fe, 0xc2974eb4ee658829},
+ {0xce947a3da6a9273e, 0x733d226229feea33},
+ {0x811ccc668829b887, 0x0806357d5a3f5260},
+ {0xa163ff802a3426a8, 0xca07c2dcb0cf26f8},
+ {0xc9bcff6034c13052, 0xfc89b393dd02f0b6},
+ {0xfc2c3f3841f17c67, 0xbbac2078d443ace3},
+ {0x9d9ba7832936edc0, 0xd54b944b84aa4c0e},
+ {0xc5029163f384a931, 0x0a9e795e65d4df12},
+ {0xf64335bcf065d37d, 0x4d4617b5ff4a16d6},
+ {0x99ea0196163fa42e, 0x504bced1bf8e4e46},
+ {0xc06481fb9bcf8d39, 0xe45ec2862f71e1d7},
+ {0xf07da27a82c37088, 0x5d767327bb4e5a4d},
+ {0x964e858c91ba2655, 0x3a6a07f8d510f870},
+ {0xbbe226efb628afea, 0x890489f70a55368c},
+ {0xeadab0aba3b2dbe5, 0x2b45ac74ccea842f},
+ {0x92c8ae6b464fc96f, 0x3b0b8bc90012929e},
+ {0xb77ada0617e3bbcb, 0x09ce6ebb40173745},
+ {0xe55990879ddcaabd, 0xcc420a6a101d0516},
+ {0x8f57fa54c2a9eab6, 0x9fa946824a12232e},
+ {0xb32df8e9f3546564, 0x47939822dc96abfa},
+ {0xdff9772470297ebd, 0x59787e2b93bc56f8},
+ {0x8bfbea76c619ef36, 0x57eb4edb3c55b65b},
+ {0xaefae51477a06b03, 0xede622920b6b23f2},
+ {0xdab99e59958885c4, 0xe95fab368e45ecee},
+ {0x88b402f7fd75539b, 0x11dbcb0218ebb415},
+ {0xaae103b5fcd2a881, 0xd652bdc29f26a11a},
+ {0xd59944a37c0752a2, 0x4be76d3346f04960},
+ {0x857fcae62d8493a5, 0x6f70a4400c562ddc},
+ {0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb953},
+ {0xd097ad07a71f26b2, 0x7e2000a41346a7a8},
+ {0x825ecc24c873782f, 0x8ed400668c0c28c9},
+ {0xa2f67f2dfa90563b, 0x728900802f0f32fb},
+ {0xcbb41ef979346bca, 0x4f2b40a03ad2ffba},
+ {0xfea126b7d78186bc, 0xe2f610c84987bfa9},
+ {0x9f24b832e6b0f436, 0x0dd9ca7d2df4d7ca},
+ {0xc6ede63fa05d3143, 0x91503d1c79720dbc},
+ {0xf8a95fcf88747d94, 0x75a44c6397ce912b},
+ {0x9b69dbe1b548ce7c, 0xc986afbe3ee11abb},
+ {0xc24452da229b021b, 0xfbe85badce996169},
+ {0xf2d56790ab41c2a2, 0xfae27299423fb9c4},
+ {0x97c560ba6b0919a5, 0xdccd879fc967d41b},
+ {0xbdb6b8e905cb600f, 0x5400e987bbc1c921},
+ {0xed246723473e3813, 0x290123e9aab23b69},
+ {0x9436c0760c86e30b, 0xf9a0b6720aaf6522},
+ {0xb94470938fa89bce, 0xf808e40e8d5b3e6a},
+ {0xe7958cb87392c2c2, 0xb60b1d1230b20e05},
+ {0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c3},
+ {0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af4},
+ {0xe2280b6c20dd5232, 0x25c6da63c38de1b1},
+ {0x8d590723948a535f, 0x579c487e5a38ad0f},
+ {0xb0af48ec79ace837, 0x2d835a9df0c6d852},
+ {0xdcdb1b2798182244, 0xf8e431456cf88e66},
+ {0x8a08f0f8bf0f156b, 0x1b8e9ecb641b5900},
+ {0xac8b2d36eed2dac5, 0xe272467e3d222f40},
+ {0xd7adf884aa879177, 0x5b0ed81dcc6abb10},
+ {0x86ccbb52ea94baea, 0x98e947129fc2b4ea},
+ {0xa87fea27a539e9a5, 0x3f2398d747b36225},
+ {0xd29fe4b18e88640e, 0x8eec7f0d19a03aae},
+ {0x83a3eeeef9153e89, 0x1953cf68300424ad},
+ {0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd8},
+ {0xcdb02555653131b6, 0x3792f412cb06794e},
+ {0x808e17555f3ebf11, 0xe2bbd88bbee40bd1},
+ {0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec5},
+ {0xc8de047564d20a8b, 0xf245825a5a445276},
+ {0xfb158592be068d2e, 0xeed6e2f0f0d56713},
+ {0x9ced737bb6c4183d, 0x55464dd69685606c},
+ {0xc428d05aa4751e4c, 0xaa97e14c3c26b887},
+ {0xf53304714d9265df, 0xd53dd99f4b3066a9},
+ {0x993fe2c6d07b7fab, 0xe546a8038efe402a},
+ {0xbf8fdb78849a5f96, 0xde98520472bdd034},
+ {0xef73d256a5c0f77c, 0x963e66858f6d4441},
+ {0x95a8637627989aad, 0xdde7001379a44aa9},
+ {0xbb127c53b17ec159, 0x5560c018580d5d53},
+ {0xe9d71b689dde71af, 0xaab8f01e6e10b4a7},
+ {0x9226712162ab070d, 0xcab3961304ca70e9},
+ {0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d23},
+ {0xe45c10c42a2b3b05, 0x8cb89a7db77c506b},
+ {0x8eb98a7a9a5b04e3, 0x77f3608e92adb243},
+ {0xb267ed1940f1c61c, 0x55f038b237591ed4},
+ {0xdf01e85f912e37a3, 0x6b6c46dec52f6689},
+ {0x8b61313bbabce2c6, 0x2323ac4b3b3da016},
+ {0xae397d8aa96c1b77, 0xabec975e0a0d081b},
+ {0xd9c7dced53c72255, 0x96e7bd358c904a22},
+ {0x881cea14545c7575, 0x7e50d64177da2e55},
+ {0xaa242499697392d2, 0xdde50bd1d5d0b9ea},
+ {0xd4ad2dbfc3d07787, 0x955e4ec64b44e865},
+ {0x84ec3c97da624ab4, 0xbd5af13bef0b113f},
+ {0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58f},
+ {0xcfb11ead453994ba, 0x67de18eda5814af3},
+ {0x81ceb32c4b43fcf4, 0x80eacf948770ced8},
+ {0xa2425ff75e14fc31, 0xa1258379a94d028e},
+ {0xcad2f7f5359a3b3e, 0x096ee45813a04331},
+ {0xfd87b5f28300ca0d, 0x8bca9d6e188853fd},
+ {0x9e74d1b791e07e48, 0x775ea264cf55347e},
+ {0xc612062576589dda, 0x95364afe032a819e},
+ {0xf79687aed3eec551, 0x3a83ddbd83f52205},
+ {0x9abe14cd44753b52, 0xc4926a9672793543},
+ {0xc16d9a0095928a27, 0x75b7053c0f178294},
+ {0xf1c90080baf72cb1, 0x5324c68b12dd6339},
+ {0x971da05074da7bee, 0xd3f6fc16ebca5e04},
+ {0xbce5086492111aea, 0x88f4bb1ca6bcf585},
+ {0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6},
+ {0x9392ee8e921d5d07, 0x3aff322e62439fd0},
+ {0xb877aa3236a4b449, 0x09befeb9fad487c3},
+ {0xe69594bec44de15b, 0x4c2ebe687989a9b4},
+ {0x901d7cf73ab0acd9, 0x0f9d37014bf60a11},
+ {0xb424dc35095cd80f, 0x538484c19ef38c95},
+ {0xe12e13424bb40e13, 0x2865a5f206b06fba},
+ {0x8cbccc096f5088cb, 0xf93f87b7442e45d4},
+ {0xafebff0bcb24aafe, 0xf78f69a51539d749},
+ {0xdbe6fecebdedd5be, 0xb573440e5a884d1c},
+ {0x89705f4136b4a597, 0x31680a88f8953031},
+ {0xabcc77118461cefc, 0xfdc20d2b36ba7c3e},
+ {0xd6bf94d5e57a42bc, 0x3d32907604691b4d},
+ {0x8637bd05af6c69b5, 0xa63f9a49c2c1b110},
+ {0xa7c5ac471b478423, 0x0fcf80dc33721d54},
+ {0xd1b71758e219652b, 0xd3c36113404ea4a9},
+ {0x83126e978d4fdf3b, 0x645a1cac083126ea},
+ {0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4},
+ {0xcccccccccccccccc, 0xcccccccccccccccd},
+ {0x8000000000000000, 0x0000000000000000},
+ {0xa000000000000000, 0x0000000000000000},
+ {0xc800000000000000, 0x0000000000000000},
+ {0xfa00000000000000, 0x0000000000000000},
+ {0x9c40000000000000, 0x0000000000000000},
+ {0xc350000000000000, 0x0000000000000000},
+ {0xf424000000000000, 0x0000000000000000},
+ {0x9896800000000000, 0x0000000000000000},
+ {0xbebc200000000000, 0x0000000000000000},
+ {0xee6b280000000000, 0x0000000000000000},
+ {0x9502f90000000000, 0x0000000000000000},
+ {0xba43b74000000000, 0x0000000000000000},
+ {0xe8d4a51000000000, 0x0000000000000000},
+ {0x9184e72a00000000, 0x0000000000000000},
+ {0xb5e620f480000000, 0x0000000000000000},
+ {0xe35fa931a0000000, 0x0000000000000000},
+ {0x8e1bc9bf04000000, 0x0000000000000000},
+ {0xb1a2bc2ec5000000, 0x0000000000000000},
+ {0xde0b6b3a76400000, 0x0000000000000000},
+ {0x8ac7230489e80000, 0x0000000000000000},
+ {0xad78ebc5ac620000, 0x0000000000000000},
+ {0xd8d726b7177a8000, 0x0000000000000000},
+ {0x878678326eac9000, 0x0000000000000000},
+ {0xa968163f0a57b400, 0x0000000000000000},
+ {0xd3c21bcecceda100, 0x0000000000000000},
+ {0x84595161401484a0, 0x0000000000000000},
+ {0xa56fa5b99019a5c8, 0x0000000000000000},
+ {0xcecb8f27f4200f3a, 0x0000000000000000},
+ {0x813f3978f8940984, 0x4000000000000000},
+ {0xa18f07d736b90be5, 0x5000000000000000},
+ {0xc9f2c9cd04674ede, 0xa400000000000000},
+ {0xfc6f7c4045812296, 0x4d00000000000000},
+ {0x9dc5ada82b70b59d, 0xf020000000000000},
+ {0xc5371912364ce305, 0x6c28000000000000},
+ {0xf684df56c3e01bc6, 0xc732000000000000},
+ {0x9a130b963a6c115c, 0x3c7f400000000000},
+ {0xc097ce7bc90715b3, 0x4b9f100000000000},
+ {0xf0bdc21abb48db20, 0x1e86d40000000000},
+ {0x96769950b50d88f4, 0x1314448000000000},
+ {0xbc143fa4e250eb31, 0x17d955a000000000},
+ {0xeb194f8e1ae525fd, 0x5dcfab0800000000},
+ {0x92efd1b8d0cf37be, 0x5aa1cae500000000},
+ {0xb7abc627050305ad, 0xf14a3d9e40000000},
+ {0xe596b7b0c643c719, 0x6d9ccd05d0000000},
+ {0x8f7e32ce7bea5c6f, 0xe4820023a2000000},
+ {0xb35dbf821ae4f38b, 0xdda2802c8a800000},
+ {0xe0352f62a19e306e, 0xd50b2037ad200000},
+ {0x8c213d9da502de45, 0x4526f422cc340000},
+ {0xaf298d050e4395d6, 0x9670b12b7f410000},
+ {0xdaf3f04651d47b4c, 0x3c0cdd765f114000},
+ {0x88d8762bf324cd0f, 0xa5880a69fb6ac800},
+ {0xab0e93b6efee0053, 0x8eea0d047a457a00},
+ {0xd5d238a4abe98068, 0x72a4904598d6d880},
+ {0x85a36366eb71f041, 0x47a6da2b7f864750},
+ {0xa70c3c40a64e6c51, 0x999090b65f67d924},
+ {0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d},
+ {0x82818f1281ed449f, 0xbff8f10e7a8921a5},
+ {0xa321f2d7226895c7, 0xaff72d52192b6a0e},
+ {0xcbea6f8ceb02bb39, 0x9bf4f8a69f764491},
+ {0xfee50b7025c36a08, 0x02f236d04753d5b5},
+ {0x9f4f2726179a2245, 0x01d762422c946591},
+ {0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef6},
+ {0xf8ebad2b84e0d58b, 0xd2e0898765a7deb3},
+ {0x9b934c3b330c8577, 0x63cc55f49f88eb30},
+ {0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fc},
+ {0xf316271c7fc3908a, 0x8bef464e3945ef7b},
+ {0x97edd871cfda3a56, 0x97758bf0e3cbb5ad},
+ {0xbde94e8e43d0c8ec, 0x3d52eeed1cbea318},
+ {0xed63a231d4c4fb27, 0x4ca7aaa863ee4bde},
+ {0x945e455f24fb1cf8, 0x8fe8caa93e74ef6b},
+ {0xb975d6b6ee39e436, 0xb3e2fd538e122b45},
+ {0xe7d34c64a9c85d44, 0x60dbbca87196b617},
+ {0x90e40fbeea1d3a4a, 0xbc8955e946fe31ce},
+ {0xb51d13aea4a488dd, 0x6babab6398bdbe42},
+ {0xe264589a4dcdab14, 0xc696963c7eed2dd2},
+ {0x8d7eb76070a08aec, 0xfc1e1de5cf543ca3},
+ {0xb0de65388cc8ada8, 0x3b25a55f43294bcc},
+ {0xdd15fe86affad912, 0x49ef0eb713f39ebf},
+ {0x8a2dbf142dfcc7ab, 0x6e3569326c784338},
+ {0xacb92ed9397bf996, 0x49c2c37f07965405},
+ {0xd7e77a8f87daf7fb, 0xdc33745ec97be907},
+ {0x86f0ac99b4e8dafd, 0x69a028bb3ded71a4},
+ {0xa8acd7c0222311bc, 0xc40832ea0d68ce0d},
+ {0xd2d80db02aabd62b, 0xf50a3fa490c30191},
+ {0x83c7088e1aab65db, 0x792667c6da79e0fb},
+ {0xa4b8cab1a1563f52, 0x577001b891185939},
+ {0xcde6fd5e09abcf26, 0xed4c0226b55e6f87},
+ {0x80b05e5ac60b6178, 0x544f8158315b05b5},
+ {0xa0dc75f1778e39d6, 0x696361ae3db1c722},
+ {0xc913936dd571c84c, 0x03bc3a19cd1e38ea},
+ {0xfb5878494ace3a5f, 0x04ab48a04065c724},
+ {0x9d174b2dcec0e47b, 0x62eb0d64283f9c77},
+ {0xc45d1df942711d9a, 0x3ba5d0bd324f8395},
+ {0xf5746577930d6500, 0xca8f44ec7ee3647a},
+ {0x9968bf6abbe85f20, 0x7e998b13cf4e1ecc},
+ {0xbfc2ef456ae276e8, 0x9e3fedd8c321a67f},
+ {0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101f},
+ {0x95d04aee3b80ece5, 0xbba1f1d158724a13},
+ {0xbb445da9ca61281f, 0x2a8a6e45ae8edc98},
+ {0xea1575143cf97226, 0xf52d09d71a3293be},
+ {0x924d692ca61be758, 0x593c2626705f9c57},
+ {0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836d},
+ {0xe498f455c38b997a, 0x0b6dfb9c0f956448},
+ {0x8edf98b59a373fec, 0x4724bd4189bd5ead},
+ {0xb2977ee300c50fe7, 0x58edec91ec2cb658},
+ {0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ee},
+ {0x8b865b215899f46c, 0xbd79e0d20082ee75},
+ {0xae67f1e9aec07187, 0xecd8590680a3aa12},
+ {0xda01ee641a708de9, 0xe80e6f4820cc9496},
+ {0x884134fe908658b2, 0x3109058d147fdcde},
+ {0xaa51823e34a7eede, 0xbd4b46f0599fd416},
+ {0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91b},
+ {0x850fadc09923329e, 0x03e2cf6bc604ddb1},
+ {0xa6539930bf6bff45, 0x84db8346b786151d},
+ {0xcfe87f7cef46ff16, 0xe612641865679a64},
+ {0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07f},
+ {0xa26da3999aef7749, 0xe3be5e330f38f09e},
+ {0xcb090c8001ab551c, 0x5cadf5bfd3072cc6},
+ {0xfdcb4fa002162a63, 0x73d9732fc7c8f7f7},
+ {0x9e9f11c4014dda7e, 0x2867e7fddcdd9afb},
+ {0xc646d63501a1511d, 0xb281e1fd541501b9},
+ {0xf7d88bc24209a565, 0x1f225a7ca91a4227},
+ {0x9ae757596946075f, 0x3375788de9b06959},
+ {0xc1a12d2fc3978937, 0x0052d6b1641c83af},
+ {0xf209787bb47d6b84, 0xc0678c5dbd23a49b},
+ {0x9745eb4d50ce6332, 0xf840b7ba963646e1},
+ {0xbd176620a501fbff, 0xb650e5a93bc3d899},
+ {0xec5d3fa8ce427aff, 0xa3e51f138ab4cebf},
+ {0x93ba47c980e98cdf, 0xc66f336c36b10138},
+ {0xb8a8d9bbe123f017, 0xb80b0047445d4185},
+ {0xe6d3102ad96cec1d, 0xa60dc059157491e6},
+ {0x9043ea1ac7e41392, 0x87c89837ad68db30},
+ {0xb454e4a179dd1877, 0x29babe4598c311fc},
+ {0xe16a1dc9d8545e94, 0xf4296dd6fef3d67b},
+ {0x8ce2529e2734bb1d, 0x1899e4a65f58660d},
+ {0xb01ae745b101e9e4, 0x5ec05dcff72e7f90},
+ {0xdc21a1171d42645d, 0x76707543f4fa1f74},
+ {0x899504ae72497eba, 0x6a06494a791c53a9},
+ {0xabfa45da0edbde69, 0x0487db9d17636893},
+ {0xd6f8d7509292d603, 0x45a9d2845d3c42b7},
+ {0x865b86925b9bc5c2, 0x0b8a2392ba45a9b3},
+ {0xa7f26836f282b732, 0x8e6cac7768d7141f},
+ {0xd1ef0244af2364ff, 0x3207d795430cd927},
+ {0x8335616aed761f1f, 0x7f44e6bd49e807b9},
+ {0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a7},
+ {0xcd036837130890a1, 0x36dba887c37a8c10},
+ {0x802221226be55a64, 0xc2494954da2c978a},
+ {0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6d},
+ {0xc83553c5c8965d3d, 0x6f92829494e5acc8},
+ {0xfa42a8b73abbf48c, 0xcb772339ba1f17fa},
+ {0x9c69a97284b578d7, 0xff2a760414536efc},
+ {0xc38413cf25e2d70d, 0xfef5138519684abb},
+ {0xf46518c2ef5b8cd1, 0x7eb258665fc25d6a},
+ {0x98bf2f79d5993802, 0xef2f773ffbd97a62},
+ {0xbeeefb584aff8603, 0xaafb550ffacfd8fb},
+ {0xeeaaba2e5dbf6784, 0x95ba2a53f983cf39},
+ {0x952ab45cfa97a0b2, 0xdd945a747bf26184},
+ {0xba756174393d88df, 0x94f971119aeef9e5},
+ {0xe912b9d1478ceb17, 0x7a37cd5601aab85e},
+ {0x91abb422ccb812ee, 0xac62e055c10ab33b},
+ {0xb616a12b7fe617aa, 0x577b986b314d600a},
+ {0xe39c49765fdf9d94, 0xed5a7e85fda0b80c},
+ {0x8e41ade9fbebc27d, 0x14588f13be847308},
+ {0xb1d219647ae6b31c, 0x596eb2d8ae258fc9},
+ {0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bc},
+ {0x8aec23d680043bee, 0x25de7bb9480d5855},
+ {0xada72ccc20054ae9, 0xaf561aa79a10ae6b},
+ {0xd910f7ff28069da4, 0x1b2ba1518094da05},
+ {0x87aa9aff79042286, 0x90fb44d2f05d0843},
+ {0xa99541bf57452b28, 0x353a1607ac744a54},
+ {0xd3fa922f2d1675f2, 0x42889b8997915ce9},
+ {0x847c9b5d7c2e09b7, 0x69956135febada12},
+ {0xa59bc234db398c25, 0x43fab9837e699096},
+ {0xcf02b2c21207ef2e, 0x94f967e45e03f4bc},
+ {0x8161afb94b44f57d, 0x1d1be0eebac278f6},
+ {0xa1ba1ba79e1632dc, 0x6462d92a69731733},
+ {0xca28a291859bbf93, 0x7d7b8f7503cfdcff},
+ {0xfcb2cb35e702af78, 0x5cda735244c3d43f},
+ {0x9defbf01b061adab, 0x3a0888136afa64a8},
+ {0xc56baec21c7a1916, 0x088aaa1845b8fdd1},
+ {0xf6c69a72a3989f5b, 0x8aad549e57273d46},
+ {0x9a3c2087a63f6399, 0x36ac54e2f678864c},
+ {0xc0cb28a98fcf3c7f, 0x84576a1bb416a7de},
+ {0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d6},
+ {0x969eb7c47859e743, 0x9f644ae5a4b1b326},
+ {0xbc4665b596706114, 0x873d5d9f0dde1fef},
+ {0xeb57ff22fc0c7959, 0xa90cb506d155a7eb},
+ {0x9316ff75dd87cbd8, 0x09a7f12442d588f3},
+ {0xb7dcbf5354e9bece, 0x0c11ed6d538aeb30},
+ {0xe5d3ef282a242e81, 0x8f1668c8a86da5fb},
+ {0x8fa475791a569d10, 0xf96e017d694487bd},
+ {0xb38d92d760ec4455, 0x37c981dcc395a9ad},
+ {0xe070f78d3927556a, 0x85bbe253f47b1418},
+ {0x8c469ab843b89562, 0x93956d7478ccec8f},
+ {0xaf58416654a6babb, 0x387ac8d1970027b3},
+ {0xdb2e51bfe9d0696a, 0x06997b05fcc0319f},
+ {0x88fcf317f22241e2, 0x441fece3bdf81f04},
+ {0xab3c2fddeeaad25a, 0xd527e81cad7626c4},
+ {0xd60b3bd56a5586f1, 0x8a71e223d8d3b075},
+ {0x85c7056562757456, 0xf6872d5667844e4a},
+ {0xa738c6bebb12d16c, 0xb428f8ac016561dc},
+ {0xd106f86e69d785c7, 0xe13336d701beba53},
+ {0x82a45b450226b39c, 0xecc0024661173474},
+ {0xa34d721642b06084, 0x27f002d7f95d0191},
+ {0xcc20ce9bd35c78a5, 0x31ec038df7b441f5},
+ {0xff290242c83396ce, 0x7e67047175a15272},
+ {0x9f79a169bd203e41, 0x0f0062c6e984d387},
+ {0xc75809c42c684dd1, 0x52c07b78a3e60869},
+ {0xf92e0c3537826145, 0xa7709a56ccdf8a83},
+ {0x9bbcc7a142b17ccb, 0x88a66076400bb692},
+ {0xc2abf989935ddbfe, 0x6acff893d00ea436},
+ {0xf356f7ebf83552fe, 0x0583f6b8c4124d44},
+ {0x98165af37b2153de, 0xc3727a337a8b704b},
+ {0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5d},
+ {0xeda2ee1c7064130c, 0x1162def06f79df74},
+ {0x9485d4d1c63e8be7, 0x8addcb5645ac2ba9},
+ {0xb9a74a0637ce2ee1, 0x6d953e2bd7173693},
+ {0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0438},
+ {0x910ab1d4db9914a0, 0x1d9c9892400a22a3},
+ {0xb54d5e4a127f59c8, 0x2503beb6d00cab4c},
+ {0xe2a0b5dc971f303a, 0x2e44ae64840fd61e},
+ {0x8da471a9de737e24, 0x5ceaecfed289e5d3},
+ {0xb10d8e1456105dad, 0x7425a83e872c5f48},
+ {0xdd50f1996b947518, 0xd12f124e28f7771a},
+ {0x8a5296ffe33cc92f, 0x82bd6b70d99aaa70},
+ {0xace73cbfdc0bfb7b, 0x636cc64d1001550c},
+ {0xd8210befd30efa5a, 0x3c47f7e05401aa4f},
+ {0x8714a775e3e95c78, 0x65acfaec34810a72},
+ {0xa8d9d1535ce3b396, 0x7f1839a741a14d0e},
+ {0xd31045a8341ca07c, 0x1ede48111209a051},
+ {0x83ea2b892091e44d, 0x934aed0aab460433},
+ {0xa4e4b66b68b65d60, 0xf81da84d56178540},
+ {0xce1de40642e3f4b9, 0x36251260ab9d668f},
+ {0x80d2ae83e9ce78f3, 0xc1d72b7c6b42601a},
+ {0xa1075a24e4421730, 0xb24cf65b8612f820},
+ {0xc94930ae1d529cfc, 0xdee033f26797b628},
+ {0xfb9b7cd9a4a7443c, 0x169840ef017da3b2},
+ {0x9d412e0806e88aa5, 0x8e1f289560ee864f},
+ {0xc491798a08a2ad4e, 0xf1a6f2bab92a27e3},
+ {0xf5b5d7ec8acb58a2, 0xae10af696774b1dc},
+ {0x9991a6f3d6bf1765, 0xacca6da1e0a8ef2a},
+ {0xbff610b0cc6edd3f, 0x17fd090a58d32af4},
+ {0xeff394dcff8a948e, 0xddfc4b4cef07f5b1},
+ {0x95f83d0a1fb69cd9, 0x4abdaf101564f98f},
+ {0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f2},
+ {0xea53df5fd18d5513, 0x84c86189216dc5ee},
+ {0x92746b9be2f8552c, 0x32fd3cf5b4e49bb5},
+ {0xb7118682dbb66a77, 0x3fbc8c33221dc2a2},
+ {0xe4d5e82392a40515, 0x0fabaf3feaa5334b},
+ {0x8f05b1163ba6832d, 0x29cb4d87f2a7400f},
+ {0xb2c71d5bca9023f8, 0x743e20e9ef511013},
+ {0xdf78e4b2bd342cf6, 0x914da9246b255417},
+ {0x8bab8eefb6409c1a, 0x1ad089b6c2f7548f},
+ {0xae9672aba3d0c320, 0xa184ac2473b529b2},
+ {0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741f},
+ {0x8865899617fb1871, 0x7e2fa67c7a658893},
+ {0xaa7eebfb9df9de8d, 0xddbb901b98feeab8},
+ {0xd51ea6fa85785631, 0x552a74227f3ea566},
+ {0x8533285c936b35de, 0xd53a88958f872760},
+ {0xa67ff273b8460356, 0x8a892abaf368f138},
+ {0xd01fef10a657842c, 0x2d2b7569b0432d86},
+ {0x8213f56a67f6b29b, 0x9c3b29620e29fc74},
+ {0xa298f2c501f45f42, 0x8349f3ba91b47b90},
+ {0xcb3f2f7642717713, 0x241c70a936219a74},
+ {0xfe0efb53d30dd4d7, 0xed238cd383aa0111},
+ {0x9ec95d1463e8a506, 0xf4363804324a40ab},
+ {0xc67bb4597ce2ce48, 0xb143c6053edcd0d6},
+ {0xf81aa16fdc1b81da, 0xdd94b7868e94050b},
+ {0x9b10a4e5e9913128, 0xca7cf2b4191c8327},
+ {0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f1},
+ {0xf24a01a73cf2dccf, 0xbc633b39673c8ced},
+ {0x976e41088617ca01, 0xd5be0503e085d814},
+ {0xbd49d14aa79dbc82, 0x4b2d8644d8a74e19},
+ {0xec9c459d51852ba2, 0xddf8e7d60ed1219f},
+ {0x93e1ab8252f33b45, 0xcabb90e5c942b504},
+ {0xb8da1662e7b00a17, 0x3d6a751f3b936244},
+ {0xe7109bfba19c0c9d, 0x0cc512670a783ad5},
+ {0x906a617d450187e2, 0x27fb2b80668b24c6},
+ {0xb484f9dc9641e9da, 0xb1f9f660802dedf7},
+ {0xe1a63853bbd26451, 0x5e7873f8a0396974},
+ {0x8d07e33455637eb2, 0xdb0b487b6423e1e9},
+ {0xb049dc016abc5e5f, 0x91ce1a9a3d2cda63},
+ {0xdc5c5301c56b75f7, 0x7641a140cc7810fc},
+ {0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9e},
+ {0xac2820d9623bf429, 0x546345fa9fbdcd45},
+ {0xd732290fbacaf133, 0xa97c177947ad4096},
+ {0x867f59a9d4bed6c0, 0x49ed8eabcccc485e},
+ {0xa81f301449ee8c70, 0x5c68f256bfff5a75},
+ {0xd226fc195c6a2f8c, 0x73832eec6fff3112},
+ {0x83585d8fd9c25db7, 0xc831fd53c5ff7eac},
+ {0xa42e74f3d032f525, 0xba3e7ca8b77f5e56},
+ {0xcd3a1230c43fb26f, 0x28ce1bd2e55f35ec},
+ {0x80444b5e7aa7cf85, 0x7980d163cf5b81b4},
+ {0xa0555e361951c366, 0xd7e105bcc3326220},
+ {0xc86ab5c39fa63440, 0x8dd9472bf3fefaa8},
+ {0xfa856334878fc150, 0xb14f98f6f0feb952},
+ {0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d4},
+ {0xc3b8358109e84f07, 0x0a862f80ec4700c9},
+ {0xf4a642e14c6262c8, 0xcd27bb612758c0fb},
+ {0x98e7e9cccfbd7dbd, 0x8038d51cb897789d},
+ {0xbf21e44003acdd2c, 0xe0470a63e6bd56c4},
+ {0xeeea5d5004981478, 0x1858ccfce06cac75},
+ {0x95527a5202df0ccb, 0x0f37801e0c43ebc9},
+ {0xbaa718e68396cffd, 0xd30560258f54e6bb},
+ {0xe950df20247c83fd, 0x47c6b82ef32a206a},
+ {0x91d28b7416cdd27e, 0x4cdc331d57fa5442},
+ {0xb6472e511c81471d, 0xe0133fe4adf8e953},
+ {0xe3d8f9e563a198e5, 0x58180fddd97723a7},
+ {0x8e679c2f5e44ff8f, 0x570f09eaa7ea7649},
+ {0xb201833b35d63f73, 0x2cd2cc6551e513db},
+ {0xde81e40a034bcf4f, 0xf8077f7ea65e58d2},
+ {0x8b112e86420f6191, 0xfb04afaf27faf783},
+ {0xadd57a27d29339f6, 0x79c5db9af1f9b564},
+ {0xd94ad8b1c7380874, 0x18375281ae7822bd},
+ {0x87cec76f1c830548, 0x8f2293910d0b15b6},
+ {0xa9c2794ae3a3c69a, 0xb2eb3875504ddb23},
+ {0xd433179d9c8cb841, 0x5fa60692a46151ec},
+ {0x849feec281d7f328, 0xdbc7c41ba6bcd334},
+ {0xa5c7ea73224deff3, 0x12b9b522906c0801},
+ {0xcf39e50feae16bef, 0xd768226b34870a01},
+ {0x81842f29f2cce375, 0xe6a1158300d46641},
+ {0xa1e53af46f801c53, 0x60495ae3c1097fd1},
+ {0xca5e89b18b602368, 0x385bb19cb14bdfc5},
+ {0xfcf62c1dee382c42, 0x46729e03dd9ed7b6},
+ {0x9e19db92b4e31ba9, 0x6c07a2c26a8346d2},
+ {0xc5a05277621be293, 0xc7098b7305241886},
+ {0xf70867153aa2db38, 0xb8cbee4fc66d1ea8}
+};
-struct cached_power // c = f * 2^e ~= 10^k
-{
- std::uint64_t f;
- int e;
- int k;
+// Per-format helper routines (binary64 specializations of the Dragonbox steps).
+struct compute_mul_result {
+ std::uint64_t integer_part;
+ bool is_integer;
+};
+struct compute_mul_parity_result {
+ bool parity;
+ bool is_integer;
};
-/*!
-For a normalized diyfp w = f * 2^e, this function returns a (normalized) cached
-power-of-ten c = f_c * 2^e_c, such that the exponent of the product w * c
-satisfies (Definition 3.2 from [1])
- alpha <= e_c + e + q <= gamma.
-*/
-inline cached_power get_cached_power_for_binary_exponent(int e) {
- // Now
- //
- // alpha <= e_c + e + q <= gamma (1)
- // ==> f_c * 2^alpha <= c * 2^e * 2^q
- //
- // and since the c's are normalized, 2^(q-1) <= f_c,
- //
- // ==> 2^(q - 1 + alpha) <= c * 2^(e + q)
- // ==> 2^(alpha - e - 1) <= c
- //
- // If c were an exact power of ten, i.e. c = 10^k, one may determine k as
- //
- // k = ceil( log_10( 2^(alpha - e - 1) ) )
- // = ceil( (alpha - e - 1) * log_10(2) )
- //
- // From the paper:
- // "In theory the result of the procedure could be wrong since c is rounded,
- // and the computation itself is approximated [...]. In practice, however,
- // this simple function is sufficient."
- //
- // For IEEE double precision floating-point numbers converted into
- // normalized diyfp's w = f * 2^e, with q = 64,
- //
- // e >= -1022 (min IEEE exponent)
- // -52 (p - 1)
- // -52 (p - 1, possibly normalize denormal IEEE numbers)
- // -11 (normalize the diyfp)
- // = -1137
- //
- // and
- //
- // e <= +1023 (max IEEE exponent)
- // -52 (p - 1)
- // -11 (normalize the diyfp)
- // = 960
- //
- // This binary exponent range [-1137,960] results in a decimal exponent
- // range [-307,324]. One does not need to store a cached power for each
- // k in this range. For each such k it suffices to find a cached power
- // such that the exponent of the product lies in [alpha,gamma].
- // This implies that the difference of the decimal exponents of adjacent
- // table entries must be less than or equal to
- //
- // floor( (gamma - alpha) * log_10(2) ) = 8.
- //
- // (A smaller distance gamma-alpha would require a larger table.)
-
- // NB:
- // Actually this function returns c, such that -60 <= e_c + e + 64 <= -34.
-
- constexpr int kCachedPowersMinDecExp = -300;
- constexpr int kCachedPowersDecStep = 8;
-
- static constexpr std::array<cached_power, 79> kCachedPowers = {{
- {0xAB70FE17C79AC6CA, -1060, -300}, {0xFF77B1FCBEBCDC4F, -1034, -292},
- {0xBE5691EF416BD60C, -1007, -284}, {0x8DD01FAD907FFC3C, -980, -276},
- {0xD3515C2831559A83, -954, -268}, {0x9D71AC8FADA6C9B5, -927, -260},
- {0xEA9C227723EE8BCB, -901, -252}, {0xAECC49914078536D, -874, -244},
- {0x823C12795DB6CE57, -847, -236}, {0xC21094364DFB5637, -821, -228},
- {0x9096EA6F3848984F, -794, -220}, {0xD77485CB25823AC7, -768, -212},
- {0xA086CFCD97BF97F4, -741, -204}, {0xEF340A98172AACE5, -715, -196},
- {0xB23867FB2A35B28E, -688, -188}, {0x84C8D4DFD2C63F3B, -661, -180},
- {0xC5DD44271AD3CDBA, -635, -172}, {0x936B9FCEBB25C996, -608, -164},
- {0xDBAC6C247D62A584, -582, -156}, {0xA3AB66580D5FDAF6, -555, -148},
- {0xF3E2F893DEC3F126, -529, -140}, {0xB5B5ADA8AAFF80B8, -502, -132},
- {0x87625F056C7C4A8B, -475, -124}, {0xC9BCFF6034C13053, -449, -116},
- {0x964E858C91BA2655, -422, -108}, {0xDFF9772470297EBD, -396, -100},
- {0xA6DFBD9FB8E5B88F, -369, -92}, {0xF8A95FCF88747D94, -343, -84},
- {0xB94470938FA89BCF, -316, -76}, {0x8A08F0F8BF0F156B, -289, -68},
- {0xCDB02555653131B6, -263, -60}, {0x993FE2C6D07B7FAC, -236, -52},
- {0xE45C10C42A2B3B06, -210, -44}, {0xAA242499697392D3, -183, -36},
- {0xFD87B5F28300CA0E, -157, -28}, {0xBCE5086492111AEB, -130, -20},
- {0x8CBCCC096F5088CC, -103, -12}, {0xD1B71758E219652C, -77, -4},
- {0x9C40000000000000, -50, 4}, {0xE8D4A51000000000, -24, 12},
- {0xAD78EBC5AC620000, 3, 20}, {0x813F3978F8940984, 30, 28},
- {0xC097CE7BC90715B3, 56, 36}, {0x8F7E32CE7BEA5C70, 83, 44},
- {0xD5D238A4ABE98068, 109, 52}, {0x9F4F2726179A2245, 136, 60},
- {0xED63A231D4C4FB27, 162, 68}, {0xB0DE65388CC8ADA8, 189, 76},
- {0x83C7088E1AAB65DB, 216, 84}, {0xC45D1DF942711D9A, 242, 92},
- {0x924D692CA61BE758, 269, 100}, {0xDA01EE641A708DEA, 295, 108},
- {0xA26DA3999AEF774A, 322, 116}, {0xF209787BB47D6B85, 348, 124},
- {0xB454E4A179DD1877, 375, 132}, {0x865B86925B9BC5C2, 402, 140},
- {0xC83553C5C8965D3D, 428, 148}, {0x952AB45CFA97A0B3, 455, 156},
- {0xDE469FBD99A05FE3, 481, 164}, {0xA59BC234DB398C25, 508, 172},
- {0xF6C69A72A3989F5C, 534, 180}, {0xB7DCBF5354E9BECE, 561, 188},
- {0x88FCF317F22241E2, 588, 196}, {0xCC20CE9BD35C78A5, 614, 204},
- {0x98165AF37B2153DF, 641, 212}, {0xE2A0B5DC971F303A, 667, 220},
- {0xA8D9D1535CE3B396, 694, 228}, {0xFB9B7CD9A4A7443C, 720, 236},
- {0xBB764C4CA7A44410, 747, 244}, {0x8BAB8EEFB6409C1A, 774, 252},
- {0xD01FEF10A657842C, 800, 260}, {0x9B10A4E5E9913129, 827, 268},
- {0xE7109BFBA19C0C9D, 853, 276}, {0xAC2820D9623BF429, 880, 284},
- {0x80444B5E7AA7CF85, 907, 292}, {0xBF21E44003ACDD2D, 933, 300},
- {0x8E679C2F5E44FF8F, 960, 308}, {0xD433179D9C8CB841, 986, 316},
- {0x9E19DB92B4E31BA9, 1013, 324},
- }};
-
- // This computation gives exactly the same results for k as
- // k = ceil((kAlpha - e - 1) * 0.30102999566398114)
- // for |e| <= 1500, but doesn't require floating-point operations.
- // NB: log_10(2) ~= 78913 / 2^18
- const int f = kAlpha - e - 1;
- const int k = (f * 78913) / (1 << 18) + static_cast<int>(f > 0);
-
- const int index = (-kCachedPowersMinDecExp + k + (kCachedPowersDecStep - 1)) /
- kCachedPowersDecStep;
-
- const cached_power cached = kCachedPowers[static_cast<std::size_t>(index)];
-
- return cached;
+inline compute_mul_result compute_mul(std::uint64_t u, uint128 c) noexcept {
+ const uint128 r = umul192_upper128(u, c);
+ return {r.high, r.low == 0};
}
-/*!
-For n != 0, returns k, such that pow10 := 10^(k-1) <= n < 10^k.
-For n == 0, returns 1 and sets pow10 := 1.
-*/
-inline int find_largest_pow10(const std::uint32_t n, std::uint32_t &pow10) {
- // LCOV_EXCL_START
- if (n >= 1000000000) {
- pow10 = 1000000000;
- return 10;
- }
- // LCOV_EXCL_STOP
- else if (n >= 100000000) {
- pow10 = 100000000;
- return 9;
- } else if (n >= 10000000) {
- pow10 = 10000000;
- return 8;
- } else if (n >= 1000000) {
- pow10 = 1000000;
- return 7;
- } else if (n >= 100000) {
- pow10 = 100000;
- return 6;
- } else if (n >= 10000) {
- pow10 = 10000;
- return 5;
- } else if (n >= 1000) {
- pow10 = 1000;
- return 4;
- } else if (n >= 100) {
- pow10 = 100;
- return 3;
- } else if (n >= 10) {
- pow10 = 10;
- return 2;
- } else {
- pow10 = 1;
- return 1;
- }
+inline std::uint64_t compute_delta(uint128 c, int beta) noexcept {
+ return c.high >> (total_bits - 1 - beta);
}
-inline void grisu2_round(char *buf, int len, std::uint64_t dist,
- std::uint64_t delta, std::uint64_t rest,
- std::uint64_t ten_k) {
-
- // <--------------------------- delta ---->
- // <---- dist --------->
- // --------------[------------------+-------------------]--------------
- // M- w M+
- //
- // ten_k
- // <------>
- // <---- rest ---->
- // --------------[------------------+----+--------------]--------------
- // w V
- // = buf * 10^k
- //
- // ten_k represents a unit-in-the-last-place in the decimal representation
- // stored in buf.
- // Decrement buf by ten_k while this takes buf closer to w.
-
- // The tests are written in this order to avoid overflow in unsigned
- // integer arithmetic.
+inline compute_mul_parity_result compute_mul_parity(std::uint64_t two_f,
+ uint128 c, int beta) noexcept {
+ // beta is always in [1, 63] here.
+ const uint128 r = umul192_lower128(two_f, c);
+ return {((r.high >> (64 - beta)) & 1) != 0,
+ ((r.high << beta) | (r.low >> (64 - beta))) == 0};
+}
- while (rest < dist && delta - rest >= ten_k &&
- (rest + ten_k < dist || dist - rest > rest + ten_k - dist)) {
- buf[len - 1]--;
- rest += ten_k;
- }
+inline std::uint64_t
+compute_left_endpoint_for_shorter_interval_case(uint128 c, int beta) noexcept {
+ return (c.high - (c.high >> (significand_bits + 2))) >>
+ (total_bits - significand_bits - 1 - beta);
}
-/*!
-Generates V = buffer * 10^decimal_exponent, such that M- <= V <= M+.
-M- and M+ must be normalized and share the same exponent -60 <= e <= -32.
-*/
-inline void grisu2_digit_gen(char *buffer, int &length, int &decimal_exponent,
- diyfp M_minus, diyfp w, diyfp M_plus) {
- static_assert(kAlpha >= -60, "internal error");
- static_assert(kGamma <= -32, "internal error");
+inline std::uint64_t
+compute_right_endpoint_for_shorter_interval_case(uint128 c, int beta) noexcept {
+ return (c.high + (c.high >> (significand_bits + 1))) >>
+ (total_bits - significand_bits - 1 - beta);
+}
- // Generates the digits (and the exponent) of a decimal floating-point
- // number V = buffer * 10^decimal_exponent in the range [M-, M+]. The diyfp's
- // w, M- and M+ share the same exponent e, which satisfies alpha <= e <=
- // gamma.
- //
- // <--------------------------- delta ---->
- // <---- dist --------->
- // --------------[------------------+-------------------]--------------
- // M- w M+
- //
- // Grisu2 generates the digits of M+ from left to right and stops as soon as
- // V is in [M-,M+].
+inline std::uint64_t
+compute_round_up_for_shorter_interval_case(uint128 c, int beta) noexcept {
+ return ((c.high >> (total_bits - significand_bits - 2 - beta)) + 1) / 2;
+}
- std::uint64_t delta =
- diyfp::sub(M_plus, M_minus)
- .f; // (significand of (M+ - M-), implicit exponent is e)
- std::uint64_t dist =
- diyfp::sub(M_plus, w)
- .f; // (significand of (M+ - w ), implicit exponent is e)
+// floor(n / 10) for the shorter-interval right endpoint (n bounded so the
+// single multiply below is exact).
+inline std::uint64_t divide_by_pow10_1(std::uint64_t n) noexcept {
+ return umul128_upper64(n, std::uint64_t(1844674407370955162ull));
+}
- // Split M+ = f * 2^e into two parts p1 and p2 (note: e < 0):
- //
- // M+ = f * 2^e
- // = ((f div 2^-e) * 2^-e + (f mod 2^-e)) * 2^e
- // = ((p1 ) * 2^-e + (p2 )) * 2^e
- // = p1 + p2 * 2^e
+// floor(n / 1000) for the larger-divisor step (n bounded as above).
+inline std::uint64_t divide_by_pow10_3(std::uint64_t n) noexcept {
+ return umul128_upper64(n, std::uint64_t(4722366482869645214ull)) >> 8;
+}
- const diyfp one(std::uint64_t{1} << -M_plus.e, M_plus.e);
+// Returns whether n is divisible by 10^kappa (= 100) and divides n by it.
+inline bool check_divisibility_and_divide_by_pow10_kappa(std::uint64_t &n) noexcept {
+ // magic number for division by 100 (kappa == 2).
+ const std::uint32_t prod = std::uint32_t(n) * std::uint32_t(656);
+ const bool result = (prod & 0xffffu) < 656u;
+ n = std::uint64_t(prod >> 16);
+ return result;
+}
- auto p1 = static_cast<std::uint32_t>(
- M_plus.f >>
- -one.e); // p1 = f div 2^-e (Since -e >= 32, p1 fits into a 32-bit int.)
- std::uint64_t p2 = M_plus.f & (one.f - 1); // p2 = f mod 2^-e
+// Strip trailing decimal zeros from significand, bumping exponent accordingly.
+// Branchless search; constants from the Dragonbox reference.
+inline void remove_trailing_zeros(std::uint64_t &significand, int &exponent) noexcept {
+ std::uint64_t r = rotr64(significand * std::uint64_t(28999941890838049ull), 8);
+ bool b = r < std::uint64_t(184467440738ull);
+ int s = b ? 1 : 0;
+ significand = b ? r : significand;
- // 1)
- //
- // Generate the digits of the integral part p1 = d[n-1]...d[1]d[0]
+ r = rotr64(significand * std::uint64_t(182622766329724561ull), 4);
+ b = r < std::uint64_t(1844674407370956ull);
+ s = s * 2 + (b ? 1 : 0);
+ significand = b ? r : significand;
- std::uint32_t pow10;
- const int k = find_largest_pow10(p1, pow10);
+ r = rotr64(significand * std::uint64_t(10330176681277348905ull), 2);
+ b = r < std::uint64_t(184467440737095517ull);
+ s = s * 2 + (b ? 1 : 0);
+ significand = b ? r : significand;
- // 10^(k-1) <= p1 < 10^k, pow10 = 10^(k-1)
- //
- // p1 = (p1 div 10^(k-1)) * 10^(k-1) + (p1 mod 10^(k-1))
- // = (d[k-1] ) * 10^(k-1) + (p1 mod 10^(k-1))
- //
- // M+ = p1 + p2 * 2^e
- // = d[k-1] * 10^(k-1) + (p1 mod 10^(k-1)) + p2 * 2^e
- // = d[k-1] * 10^(k-1) + ((p1 mod 10^(k-1)) * 2^-e + p2) * 2^e
- // = d[k-1] * 10^(k-1) + ( rest) * 2^e
- //
- // Now generate the digits d[n] of p1 from left to right (n = k-1,...,0)
- //
- // p1 = d[k-1]...d[n] * 10^n + d[n-1]...d[0]
- //
- // but stop as soon as
- //
- // rest * 2^e = (d[n-1]...d[0] * 2^-e + p2) * 2^e <= delta * 2^e
+ r = rotr64(significand * std::uint64_t(14757395258967641293ull), 1);
+ b = r < std::uint64_t(1844674407370955162ull);
+ s = s * 2 + (b ? 1 : 0);
+ significand = b ? r : significand;
- int n = k;
- while (n > 0) {
- // Invariants:
- // M+ = buffer * 10^n + (p1 + p2 * 2^e) (buffer = 0 for n = k)
- // pow10 = 10^(n-1) <= p1 < 10^n
- //
- const std::uint32_t d = p1 / pow10; // d = p1 div 10^(n-1)
- const std::uint32_t r = p1 % pow10; // r = p1 mod 10^(n-1)
- //
- // M+ = buffer * 10^n + (d * 10^(n-1) + r) + p2 * 2^e
- // = (buffer * 10 + d) * 10^(n-1) + (r + p2 * 2^e)
- //
- buffer[length++] = static_cast<char>('0' + d); // buffer := buffer * 10 + d
- //
- // M+ = buffer * 10^(n-1) + (r + p2 * 2^e)
- //
- p1 = r;
- n--;
- //
- // M+ = buffer * 10^n + (p1 + p2 * 2^e)
- // pow10 = 10^n
- //
+ exponent += s;
+}
- // Now check if enough digits have been generated.
- // Compute
- //
- // p1 + p2 * 2^e = (p1 * 2^-e + p2) * 2^e = rest * 2^e
- //
- // Note:
- // Since rest and delta share the same exponent e, it suffices to
- // compare the significands.
- const std::uint64_t rest = (std::uint64_t{p1} << -one.e) + p2;
- if (rest <= delta) {
- // V = buffer * 10^n, with M- <= V <= M+.
+// Dragonbox core: shortest (significand, exponent) such that
+// value == significand * 10^exponent
+// for a finite, positive, non-zero binary64 value, decomposed into its raw
+// significand bits and biased exponent bits.
+struct decimal_fp {
+ std::uint64_t significand;
+ int exponent;
+};
- decimal_exponent += n;
+inline decimal_fp to_decimal(std::uint64_t binary_significand,
+ int binary_exponent) noexcept {
+ const bool is_even = (binary_significand % 2 == 0);
+ std::uint64_t two_fc = binary_significand * 2;
+
+ // Is the input a normal number?
+ if (binary_exponent != 0) {
+ binary_exponent += exponent_bias - significand_bits;
+
+ // Shorter interval case; proceed like Schubfach.
+ if (two_fc == 0) {
+ const int minus_k =
+ floor_log10_pow2_minus_log10_4_over_3(binary_exponent);
+ const int beta = binary_exponent + floor_log2_pow10(-minus_k);
+ const uint128 c = cache[-minus_k - cache_min_k];
+
+ std::uint64_t xi =
+ compute_left_endpoint_for_shorter_interval_case(c, beta);
+ const std::uint64_t zi =
+ compute_right_endpoint_for_shorter_interval_case(c, beta);
+
+ // If the left endpoint is not an integer, increase it. (Both endpoints
+ // are always included since the significand is even.)
+ if (!(binary_exponent >=
+ case_shorter_interval_left_endpoint_lower_threshold &&
+ binary_exponent <=
+ case_shorter_interval_left_endpoint_upper_threshold)) {
+ ++xi;
+ }
- // We may now just stop. But instead look if the buffer could be
- // decremented to bring V closer to w.
- //
- // pow10 = 10^n is now 1 ulp in the decimal representation V.
- // The rounding procedure works with diyfp's with an implicit
- // exponent of e.
- //
- // 10^n = (10^n * 2^-e) * 2^e = ulp * 2^e
- //
- const std::uint64_t ten_n = std::uint64_t{pow10} << -one.e;
- grisu2_round(buffer, length, dist, delta, rest, ten_n);
+ // Try the bigger divisor.
+ std::uint64_t decimal_significand = divide_by_pow10_1(zi);
+ if (decimal_significand * 10 >= xi) {
+ int decimal_exponent = minus_k + 1;
+ remove_trailing_zeros(decimal_significand, decimal_exponent);
+ return {decimal_significand, decimal_exponent};
+ }
- return;
+ // Otherwise, compute the round-up of y.
+ decimal_significand =
+ compute_round_up_for_shorter_interval_case(c, beta);
+ // On a tie, choose the even one.
+ if ((decimal_significand % 2 != 0) &&
+ binary_exponent >= shorter_interval_tie_lower_threshold &&
+ binary_exponent <= shorter_interval_tie_upper_threshold) {
+ --decimal_significand;
+ } else if (decimal_significand < xi) {
+ ++decimal_significand;
+ }
+ return {decimal_significand, minus_k};
}
- pow10 /= 10;
- //
- // pow10 = 10^(n-1) <= p1 < 10^n
- // Invariants restored.
- }
-
- // 2)
- //
- // The digits of the integral part have been generated:
- //
- // M+ = d[k-1]...d[1]d[0] + p2 * 2^e
- // = buffer + p2 * 2^e
- //
- // Now generate the digits of the fractional part p2 * 2^e.
- //
- // Note:
- // No decimal point is generated: the exponent is adjusted instead.
- //
- // p2 actually represents the fraction
- //
- // p2 * 2^e
- // = p2 / 2^-e
- // = d[-1] / 10^1 + d[-2] / 10^2 + ...
- //
- // Now generate the digits d[-m] of p1 from left to right (m = 1,2,...)
- //
- // p2 * 2^e = d[-1]d[-2]...d[-m] * 10^-m
- // + 10^-m * (d[-m-1] / 10^1 + d[-m-2] / 10^2 + ...)
- //
- // using
- //
- // 10^m * p2 = ((10^m * p2) div 2^-e) * 2^-e + ((10^m * p2) mod 2^-e)
- // = ( d) * 2^-e + ( r)
- //
- // or
- // 10^m * p2 * 2^e = d + r * 2^e
- //
- // i.e.
- //
- // M+ = buffer + p2 * 2^e
- // = buffer + 10^-m * (d + r * 2^e)
- // = (buffer * 10^m + d) * 10^-m + 10^-m * r * 2^e
- //
- // and stop as soon as 10^-m * r * 2^e <= delta * 2^e
-
- int m = 0;
- for (;;) {
- // Invariant:
- // M+ = buffer * 10^-m + 10^-m * (d[-m-1] / 10 + d[-m-2] / 10^2 + ...)
- // * 2^e
- // = buffer * 10^-m + 10^-m * (p2 )
- // * 2^e = buffer * 10^-m + 10^-m * (1/10 * (10 * p2) ) * 2^e =
- // buffer * 10^-m + 10^-m * (1/10 * ((10*p2 div 2^-e) * 2^-e +
- // (10*p2 mod 2^-e)) * 2^e
- //
- p2 *= 10;
- const std::uint64_t d = p2 >> -one.e; // d = (10 * p2) div 2^-e
- const std::uint64_t r = p2 & (one.f - 1); // r = (10 * p2) mod 2^-e
- //
- // M+ = buffer * 10^-m + 10^-m * (1/10 * (d * 2^-e + r) * 2^e
- // = buffer * 10^-m + 10^-m * (1/10 * (d + r * 2^e))
- // = (buffer * 10 + d) * 10^(-m-1) + 10^(-m-1) * r * 2^e
- //
- buffer[length++] = static_cast<char>('0' + d); // buffer := buffer * 10 + d
- //
- // M+ = buffer * 10^(-m-1) + 10^(-m-1) * r * 2^e
- //
- p2 = r;
- m++;
- //
- // M+ = buffer * 10^-m + 10^-m * p2 * 2^e
- // Invariant restored.
-
- // Check if enough digits have been generated.
- //
- // 10^-m * p2 * 2^e <= delta * 2^e
- // p2 * 2^e <= 10^m * delta * 2^e
- // p2 <= 10^m * delta
- delta *= 10;
- dist *= 10;
- if (p2 <= delta) {
+ // Normal interval case.
+ two_fc |= (std::uint64_t(1) << (significand_bits + 1));
+ } else {
+ // Subnormal number: normal interval case.
+ binary_exponent = min_exponent - significand_bits;
+ }
+
+ // Step 1: Schubfach multiplier calculation.
+ const int minus_k = floor_log10_pow2(binary_exponent) - kappa;
+ const uint128 c = cache[-minus_k - cache_min_k];
+ const int beta = binary_exponent + floor_log2_pow10(-minus_k);
+
+ const std::uint64_t deltai = compute_delta(c, beta);
+ const compute_mul_result z_result =
+ compute_mul((two_fc | 1) << beta, c);
+
+ // Step 2: Try larger divisor; remove trailing zeros if necessary.
+ std::uint64_t decimal_significand = divide_by_pow10_3(z_result.integer_part);
+ std::uint64_t r =
+ z_result.integer_part - std::uint64_t(big_divisor) * decimal_significand;
+
+ do {
+ if (r < deltai) {
+ // Exclude the right endpoint if necessary.
+ if ((r | std::uint64_t(!z_result.is_integer) | std::uint64_t(is_even)) ==
+ 0) {
+ --decimal_significand;
+ r = big_divisor;
+ break;
+ }
+ } else if (r > deltai) {
break;
+ } else {
+ // r == deltai; compare fractional parts.
+ const compute_mul_parity_result x_result =
+ compute_mul_parity(two_fc - 1, c, beta);
+ if (!(x_result.parity | (x_result.is_integer & is_even))) {
+ break;
+ }
}
- }
-
- // V = buffer * 10^-m, with M- <= V <= M+.
-
- decimal_exponent -= m;
-
- // 1 ulp in the decimal representation is now 10^-m.
- // Since delta and dist are now scaled by 10^m, we need to do the
- // same with ulp in order to keep the units in sync.
- //
- // 10^m * 10^-m = 1 = 2^-e * 2^e = ten_m * 2^e
- //
- const std::uint64_t ten_m = one.f;
- grisu2_round(buffer, length, dist, delta, p2, ten_m);
- // By construction this algorithm generates the shortest possible decimal
- // number (Loitsch, Theorem 6.2) which rounds back to w.
- // For an input number of precision p, at least
- //
- // N = 1 + ceil(p * log_10(2))
- //
- // decimal digits are sufficient to identify all binary floating-point
- // numbers (Matula, "In-and-Out conversions").
- // This implies that the algorithm does not produce more than N decimal
- // digits.
- //
- // N = 17 for p = 53 (IEEE double precision)
- // N = 9 for p = 24 (IEEE single precision)
-}
+ int decimal_exponent = minus_k + kappa + 1;
+ remove_trailing_zeros(decimal_significand, decimal_exponent);
+ return {decimal_significand, decimal_exponent};
+ } while (false);
-/*!
-v = buf * 10^decimal_exponent
-len is the length of the buffer (number of decimal digits)
-The buffer must be large enough, i.e. >= max_digits10.
-*/
-inline void grisu2(char *buf, int &len, int &decimal_exponent, diyfp m_minus,
- diyfp v, diyfp m_plus) {
-
- // --------(-----------------------+-----------------------)-------- (A)
- // m- v m+
- //
- // --------------------(-----------+-----------------------)-------- (B)
- // m- v m+
- //
- // First scale v (and m- and m+) such that the exponent is in the range
- // [alpha, gamma].
+ // Step 3: Find the significand with the smaller divisor.
+ decimal_significand *= 10;
- const cached_power cached = get_cached_power_for_binary_exponent(m_plus.e);
+ std::uint64_t dist = r - (deltai / 2) + (small_divisor / 2);
+ const bool approx_y_parity = ((dist ^ (small_divisor / 2)) & 1) != 0;
- const diyfp c_minus_k(cached.f, cached.e); // = c ~= 10^-k
+ const bool divisible_by_small_divisor =
+ check_divisibility_and_divide_by_pow10_kappa(dist);
- // The exponent of the products is = v.e + c_minus_k.e + q and is in the range
- // [alpha,gamma]
- const diyfp w = diyfp::mul(v, c_minus_k);
- const diyfp w_minus = diyfp::mul(m_minus, c_minus_k);
- const diyfp w_plus = diyfp::mul(m_plus, c_minus_k);
+ decimal_significand += dist;
- // ----(---+---)---------------(---+---)---------------(---+---)----
- // w- w w+
- // = c*m- = c*v = c*m+
- //
- // diyfp::mul rounds its result and c_minus_k is approximated too. w, w- and
- // w+ are now off by a small amount.
- // In fact:
- //
- // w - v * 10^k < 1 ulp
- //
- // To account for this inaccuracy, add resp. subtract 1 ulp.
- //
- // --------+---[---------------(---+---)---------------]---+--------
- // w- M- w M+ w+
- //
- // Now any number in [M-, M+] (bounds included) will round to w when input,
- // regardless of how the input rounding algorithm breaks ties.
- //
- // And digit_gen generates the shortest possible such number in [M-, M+].
- // Note that this does not mean that Grisu2 always generates the shortest
- // possible number in the interval (m-, m+).
- const diyfp M_minus(w_minus.f + 1, w_minus.e);
- const diyfp M_plus(w_plus.f - 1, w_plus.e);
-
- decimal_exponent = -cached.k; // = -(-k) = k
+ if (divisible_by_small_divisor) {
+ const compute_mul_parity_result y_result =
+ compute_mul_parity(two_fc, c, beta);
+ if (y_result.parity != approx_y_parity) {
+ --decimal_significand;
+ } else if ((decimal_significand % 2) != 0 && y_result.is_integer) {
+ // On a tie (y is an integer), choose the even one.
+ --decimal_significand;
+ }
+ }
- grisu2_digit_gen(buf, len, decimal_exponent, M_minus, w, M_plus);
+ return {decimal_significand, minus_k + kappa};
}
/*!
-v = buf * 10^decimal_exponent
-len is the length of the buffer (number of decimal digits)
-The buffer must be large enough, i.e. >= max_digits10.
+Fills buf with the shortest decimal digits of 'value' (which must be finite,
+positive and non-zero), sets len to the number of digits, and decimal_exponent
+so that value == (buf, interpreted as an integer) * 10^decimal_exponent.
+This mirrors the contract of the previous grisu2() entry point.
*/
-template <typename FloatType>
-void grisu2(char *buf, int &len, int &decimal_exponent, FloatType value) {
- static_assert(diyfp::kPrecision >= std::numeric_limits<FloatType>::digits + 3,
- "internal error: not enough precision");
-
- // If the neighbors (and boundaries) of 'value' are always computed for
- // double-precision numbers, all float's can be recovered using strtod (and
- // strtof). However, the resulting decimal representations are not exactly
- // "short".
- //
- // The documentation for 'std::to_chars'
- // (https://en.cppreference.com/w/cpp/utility/to_chars) says "value is
- // converted to a string as if by std::sprintf in the default ("C") locale"
- // and since sprintf promotes float's to double's, I think this is exactly
- // what 'std::to_chars' does. On the other hand, the documentation for
- // 'std::to_chars' requires that "parsing the representation using the
- // corresponding std::from_chars function recovers value exactly". That
- // indicates that single precision floating-point numbers should be recovered
- // using 'std::strtof'.
- //
- // NB: If the neighbors are computed for single-precision numbers, there is a
- // single float
- // (7.0385307e-26f) which can't be recovered using strtod. The resulting
- // double precision value is off by 1 ulp.
-#if 0
- const boundaries w = compute_boundaries(static_cast<double>(value));
-#else
- const boundaries w = compute_boundaries(value);
-#endif
-
- grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
+inline void dragonbox(char *buf, int &len, int &decimal_exponent,
+ double value) {
+ std::uint64_t bits;
+ std::memcpy(&bits, &value, sizeof(bits));
+ const std::uint64_t binary_significand =
+ bits & ((std::uint64_t(1) << significand_bits) - 1);
+ const int binary_exponent =
+ int((bits >> significand_bits) & 0x7ff);
+
+ const decimal_fp dec = to_decimal(binary_significand, binary_exponent);
+
+ // Convert the decimal significand to digits:
+ // 1) Proceed 2 digits at a time (s % 100) via a 00..99 lookup table
+ // (see Alexandrescu, "Three Optimization Tips for C++", 2012),
+ // 2) Digits come out least-significant first, writing them back-to-front
+ // with p = tmp + sizeof(tmp); to avoid reversal pass
+ // 3) Proceed remaining digits after loop to avoid branchs inside it
+ // 4) memcpy digits to char* buf (inside function input)
+ static const char digits2[201] =
+ "0001020304050607080910111213141516171819"
+ "2021222324252627282930313233343536373839"
+ "4041424344454647484950515253545556575859"
+ "6061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+ // Digit area is 24; +16 padding lets us always memcpy 16 (+1) bytes with a
+ // compile-time size so the compiler inlines (no libc size-class branches).
+ // Callers must provide to_chars_buffer_size (40) bytes for the same reason.
+ constexpr int digit_area = 24;
+ char tmp[digit_area + 16];
+ char *p = tmp + digit_area; // write backward
+ std::uint64_t s = dec.significand;
+ while (s >= 100) {
+ const std::uint32_t idx = static_cast<std::uint32_t>(s % 100) * 2;
+ s /= 100;
+ p -= 2;
+ p[0] = digits2[idx];
+ p[1] = digits2[idx + 1];
+ }
+ if (s >= 10) {
+ const std::uint32_t idx = static_cast<std::uint32_t>(s) * 2;
+ p -= 2;
+ p[0] = digits2[idx];
+ p[1] = digits2[idx + 1];
+ } else {
+ *--p = static_cast<char>('0' + s);
+ }
+ // Fixed-size copy: double has at most 17 significant digits.
+ std::memcpy(buf, p, 16);
+ buf[16] = p[16];
+ len = static_cast<int>(tmp + digit_area - p);
+ decimal_exponent = dec.exponent;
}
/*!
@@ -4111,11 +4625,16 @@ inline char *format_buffer(char *buf, int len, int decimal_exponent,
// k is the length of the buffer (number of decimal digits)
// n is the position of the decimal point relative to the start of the buffer.
+ // All mem* sizes below are compile-time constants so the compiler inlines
+ // them as plain loads/stores. That requires over-writing past the logical
+ // string length; callers must reserve to_chars_buffer_size (40) bytes.
+ // Logical output is still bounded by ~24 characters; only the returned
+ // pointer reflects the true length.
+
if (k <= n && n <= max_exp) {
// digits[000]
// len <= max_exp + 2
-
- std::memset(buf + k, '0', static_cast<size_t>(n) - static_cast<size_t>(k));
+ std::memset(buf + k, '0', 16);
// Make it look like a floating-point number (#362, #378)
buf[n + 0] = '.';
buf[n + 1] = '0';
@@ -4125,690 +4644,5307 @@ inline char *format_buffer(char *buf, int len, int decimal_exponent,
if (0 < n && n <= max_exp) {
// dig.its
// len <= max_digits10 + 1
- std::memmove(buf + (static_cast<size_t>(n) + 1), buf + n,
- static_cast<size_t>(k) - static_cast<size_t>(n));
+ // Shift the fractional digits one place right via a temp (overlap).
+ char shifted[16];
+ std::memcpy(shifted, buf + static_cast<size_t>(n), 16);
+ std::memcpy(buf + (static_cast<size_t>(n) + 1), shifted, 16);
buf[n] = '.';
return buf + (static_cast<size_t>(k) + 1U);
}
- if (min_exp < n && n <= 0) {
- // 0.[000]digits
- // len <= 2 + (-min_exp - 1) + max_digits10
+ if (min_exp < n && n <= 0) {
+ // 0.[000]digits
+ // With kMinExp = -4, n is in {-3,-2,-1,0}, so pad = -n is 0..3.
+ // len <= 2 + (-min_exp - 1) + max_digits10
+ char digits[17];
+ std::memcpy(digits, buf, 17);
+ const size_t pad = static_cast<size_t>(-n); // 0..3
+ buf[0] = '0';
+ buf[1] = '.';
+ // Fixed upper bound on leading zeros; only the first `pad` matter.
+ std::memset(buf + 2, '0', 4);
+ std::memcpy(buf + 2 + pad, digits, 17);
+ return buf + (2U + pad + static_cast<size_t>(k));
+ }
+
+ if (k == 1) {
+ // dE+123
+ // len <= 1 + 5
+ buf += 1;
+ } else {
+ // d.igitsE+123
+ // len <= max_digits10 + 1 + 5
+ // k-1 <= 16 for double; fixed-size shift via temp (overlap).
+ char shifted[16];
+ std::memcpy(shifted, buf + 1, 16);
+ std::memcpy(buf + 2, shifted, 16);
+ buf[1] = '.';
+ buf += 1 + static_cast<size_t>(k);
+ }
+
+ *buf++ = 'e';
+ return append_exponent(buf, n - 1);
+}
+
+} // NS dtoa_impl
+
+/*!
+The format of the resulting decimal representation is similar to printf's %g
+format. Returns an iterator pointing past-the-end of the decimal representation.
+@note The input number must be finite, i.e. NaN's and Inf's are not supported.
+@note The buffer must have at least to_chars_buffer_size (40) writable bytes.
+ Only ~24 characters are ever part of the logical result, but fixed-size
+ 16/17-byte mem* over-writes require the extra scratch for safety.
+@note The result is NOT null-terminated.
+*/
+char *to_chars(char *first, const char *last, double value) {
+ static_cast<void>(last); // maybe unused - fix warning
+ bool negative = std::signbit(value);
+ if (negative) {
+ value = -value;
+ *first++ = '-';
+ }
+
+ if (value == 0) // +-0
+ {
+ *first++ = '0';
+ // Make it look like a floating-point number (#362, #378)
+ *first++ = '.';
+ *first++ = '0';
+ return first;
+ }
+ // Compute v = buffer * 10^decimal_exponent.
+ // The decimal digits are stored in the buffer, which needs to be interpreted
+ // as an unsigned decimal integer.
+ // len is the length of the buffer, i.e. the number of decimal digits.
+ int len = 0;
+ int decimal_exponent = 0;
+ dtoa_impl::dragonbox(first, len, decimal_exponent, value);
+ // Format the buffer like printf("%.*g", prec, value)
+ constexpr int kMinExp = -4;
+ constexpr int kMaxExp = std::numeric_limits<double>::digits10;
+
+ return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp,
+ kMaxExp);
+}
+} // NS internal
+} // NS simdjson
+
+#endif // SIMDJSON_SRC_TO_CHARS_CPP
+
+
+/* end file to_chars.cpp */
+/* including from_chars.cpp: #include <from_chars.cpp> */
+/* begin file from_chars.cpp */
+#ifndef SIMDJSON_SRC_FROM_CHARS_CPP
+#define SIMDJSON_SRC_FROM_CHARS_CPP
+
+/* skipped duplicate #include <base.h> */
+
+/* including simdjson/internal/fast_float.h: #include "simdjson/internal/fast_float.h" */
+/* begin file simdjson/internal/fast_float.h */
+// Vendored from fast_float v8.2.10, generated by tools/vendor_fast_float.sh.
+// Do not edit by hand; re-run the script to update.
+//
+// https://github.com/fastfloat/fast_float
+// Licensed under Apache-2.0 OR MIT OR BSL-1.0, at your option.
+//
+// simdjson uses this for two things that its own number parser cannot do:
+// * the slow path for numbers with more than 19 significant digits, where
+// fast_float's bigint comparison is several times quicker than the
+// Wuffs-derived decimal shifting it replaced (see src/from_chars.cpp), and
+// * correctly rounded parsing inside a constant expression, which the runtime
+// path cannot offer because it relies on memcpy and __uint128_t (see
+// compile_time_json-inl.h).
+//
+// Two edits are applied by the script. Every fast_float name is rewritten so
+// that this copy cannot collide with a copy of fast_float that the surrounding
+// program includes for itself: namespace fast_float -> simdjson_fast_float,
+// FASTFLOAT_* -> SIMDJSON_FASTFLOAT_*, fastfloat_* -> simdjson_fastfloat_*. And
+// the accented letters in the attribution comments below are folded to ASCII,
+// to keep the tree ASCII-only; no disrespect to the people named is intended.
+// simdjson_fast_float by Daniel Lemire
+// simdjson_fast_float by Joao Paulo Magalhaes
+//
+//
+// with contributions from Eugene Golushkov
+// with contributions from Maksim Kita
+// with contributions from Marcin Wojdyr
+// with contributions from Neal Richardson
+// with contributions from Tim Paine
+// with contributions from Fabio Pellacini
+// with contributions from Lenard Szolnoki
+// with contributions from Jan Pharago
+// with contributions from Maya Warrier
+// with contributions from Taha Khokhar
+// with contributions from Anders Dalvander
+//
+//
+// Licensed under the Apache License, Version 2.0, or the
+// MIT License or the Boost License. This file may not be copied,
+// modified, or distributed except according to those terms.
+//
+// MIT License Notice
+//
+// MIT License
+//
+// Copyright (c) 2021 The simdjson_fast_float authors
+//
+// Permission is hereby granted, free of charge, to any
+// person obtaining a copy of this software and associated
+// documentation files (the "Software"), to deal in the
+// Software without restriction, including without
+// limitation the rights to use, copy, modify, merge,
+// publish, distribute, sublicense, and/or sell copies of
+// the Software, and to permit persons to whom the Software
+// is furnished to do so, subject to the following
+// conditions:
+//
+// The above copyright notice and this permission notice
+// shall be included in all copies or substantial portions
+// of the Software.
+//
+// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF
+// ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED
+// TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
+// PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT
+// SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
+// CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
+// OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR
+// IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+// DEALINGS IN THE SOFTWARE.
+//
+// Apache License (Version 2.0) Notice
+//
+// Copyright 2021 The simdjson_fast_float authors
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+//
+// BOOST License Notice
+//
+// Boost Software License - Version 1.0 - August 17th, 2003
+//
+// Permission is hereby granted, free of charge, to any person or organization
+// obtaining a copy of the software and accompanying documentation covered by
+// this license (the "Software") to use, reproduce, display, distribute,
+// execute, and transmit the Software, and to prepare derivative works of the
+// Software, and to permit third-parties to whom the Software is furnished to
+// do so, all subject to the following:
+//
+// The copyright notices in the Software and this entire statement, including
+// the above license grant, this restriction and the following disclaimer,
+// must be included in all copies of the Software, in whole or in part, and
+// all derivative works of the Software, unless such copies or derivative
+// works are solely in the form of machine-executable object code generated by
+// a source language processor.
+//
+// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+// FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
+// SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
+// FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
+// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+// DEALINGS IN THE SOFTWARE.
+//
+
+#ifndef SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+#define SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifdef __has_include
+#if __has_include(<version>)
+#include <version>
+#endif
+#endif
+
+// Testing for https://wg21.link/N3652, adopted in C++14
+#if defined(__cpp_constexpr) && __cpp_constexpr >= 201304
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14 constexpr
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14
+#endif
+
+#if defined(__cpp_lib_bit_cast) && __cpp_lib_bit_cast >= 201806L
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 0
+#endif
+
+#if defined(__cpp_lib_is_constant_evaluated) && \
+ __cpp_lib_is_constant_evaluated >= 201811L
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 0
+#endif
+
+#if defined(__cpp_if_constexpr) && __cpp_if_constexpr >= 201606L
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if constexpr (x)
+#else
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if (x)
+#endif
+
+// Testing for relevant C++20 constexpr library features
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST && \
+ defined(__cpp_lib_constexpr_algorithms) && \
+ __cpp_lib_constexpr_algorithms >= 201806L /*For std::copy and std::fill*/
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20 constexpr
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 1
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 0
+#endif
+
+#if __cplusplus >= 201703L || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L)
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 0
+#else
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 1
+#endif
+
+#endif // SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifndef SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+#define SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+
+#include <cfloat>
+#include <cstddef>
+#include <cstdint>
+#include <cassert>
+#include <cstring>
+#include <limits>
+#include <type_traits>
+#include <system_error>
+#ifdef __has_include
+#if __has_include(<stdfloat>) && (__cplusplus > 202002L || (defined(_MSVC_LANG) && (_MSVC_LANG > 202002L)))
+#include <stdfloat>
+#endif
+#endif
+
+#define SIMDJSON_FASTFLOAT_VERSION_MAJOR 8
+#define SIMDJSON_FASTFLOAT_VERSION_MINOR 2
+#define SIMDJSON_FASTFLOAT_VERSION_PATCH 10
+
+#define SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x) #x
+#define SIMDJSON_FASTFLOAT_STRINGIZE(x) SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x)
+
+#define SIMDJSON_FASTFLOAT_VERSION_STR \
+ SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MAJOR) \
+ "." SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MINOR) "." SIMDJSON_FASTFLOAT_STRINGIZE( \
+ SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+#define SIMDJSON_FASTFLOAT_VERSION \
+ (SIMDJSON_FASTFLOAT_VERSION_MAJOR * 10000 + SIMDJSON_FASTFLOAT_VERSION_MINOR * 100 + \
+ SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+namespace simdjson_fast_float {
+
+enum class chars_format : uint64_t;
+
+namespace detail {
+constexpr chars_format basic_json_fmt = chars_format(1 << 5);
+constexpr chars_format basic_fortran_fmt = chars_format(1 << 6);
+} // namespace detail
+
+enum class chars_format : uint64_t {
+ scientific = 1 << 0,
+ fixed = 1 << 2,
+ hex = 1 << 3,
+ no_infnan = 1 << 4,
+ // RFC 8259: https://datatracker.ietf.org/doc/html/rfc8259#section-6
+ json = uint64_t(detail::basic_json_fmt) | fixed | scientific | no_infnan,
+ // Extension of RFC 8259 where, e.g., "inf" and "nan" are allowed.
+ json_or_infnan = uint64_t(detail::basic_json_fmt) | fixed | scientific,
+ fortran = uint64_t(detail::basic_fortran_fmt) | fixed | scientific,
+ general = fixed | scientific,
+ allow_leading_plus = 1 << 7,
+ skip_white_space = 1 << 8,
+};
+
+template <typename UC> struct from_chars_result_t {
+ UC const *ptr;
+ std::errc ec;
+
+ // https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p2497r0.html
+ constexpr explicit operator bool() const noexcept {
+ return ec == std::errc();
+ }
+};
+
+using from_chars_result = from_chars_result_t<char>;
+
+template <typename UC> struct parse_options_t {
+ constexpr explicit parse_options_t(chars_format fmt = chars_format::general,
+ UC dot = UC('.'), int b = 10)
+ : format(fmt), decimal_point(dot), base(b) {}
+
+ /** Which number formats are accepted */
+ chars_format format;
+ /** The character used as decimal point */
+ UC decimal_point;
+ /** The base used for integers */
+ int base;
+};
+
+using parse_options = parse_options_t<char>;
+
+} // namespace simdjson_fast_float
+
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+#include <bit>
+#endif
+
+#if (defined(__x86_64) || defined(__x86_64__) || defined(_M_X64) || \
+ defined(__amd64) || defined(__aarch64__) || defined(_M_ARM64) || \
+ defined(__MINGW64__) || defined(__s390x__) || \
+ (defined(__ppc64__) || defined(__PPC64__) || defined(__ppc64le__) || \
+ defined(__PPC64LE__)) || \
+ defined(__loongarch64) || (defined(__riscv) && __riscv_xlen == 64))
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#elif (defined(__i386) || defined(__i386__) || defined(_M_IX86) || \
+ defined(__arm__) || defined(_M_ARM) || defined(__ppc__) || \
+ defined(__MINGW32__) || defined(__EMSCRIPTEN__) || \
+ (defined(__riscv) && __riscv_xlen == 32))
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#else
+ // Need to check incrementally, since SIZE_MAX is a size_t, avoid overflow.
+// We can never tell the register width, but the SIZE_MAX is a good
+// approximation. UINTPTR_MAX and INTPTR_MAX are optional, so avoid them for max
+// portability.
+#if SIZE_MAX == 0xffff
+#error Unknown platform (16-bit, unsupported)
+#elif SIZE_MAX == 0xffffffff
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#elif SIZE_MAX == 0xffffffffffffffff
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#else
+#error Unknown platform (not 32-bit, not 64-bit?)
+#endif
+#endif
+
+#if ((defined(_WIN32) || defined(_WIN64)) && !defined(__clang__)) || \
+ (defined(_M_ARM64) && !defined(__MINGW32__))
+#include <intrin.h>
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#define SIMDJSON_FASTFLOAT_VISUAL_STUDIO 1
+#endif
+
+#if defined __BYTE_ORDER__ && defined __ORDER_BIG_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+#elif defined _WIN32
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#if defined(__APPLE__) || defined(__FreeBSD__)
+#include <machine/endian.h>
+#elif defined(sun) || defined(__sun)
+#include <sys/byteorder.h>
+#elif defined(__MVS__)
+#include <sys/endian.h>
+#else
+#ifdef __has_include
+#if __has_include(<endian.h>)
+#include <endian.h>
+#endif //__has_include(<endian.h>)
+#endif //__has_include
+#endif
+#
+#ifndef __BYTE_ORDER__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#ifndef __ORDER_LITTLE_ENDIAN__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 1
+#endif
+#endif
+
+#if defined(__SSE2__) || (defined(SIMDJSON_FASTFLOAT_VISUAL_STUDIO) && \
+ (defined(_M_AMD64) || defined(_M_X64) || \
+ (defined(_M_IX86_FP) && _M_IX86_FP == 2)))
+#define SIMDJSON_FASTFLOAT_SSE2 1
+#endif
+
+#if defined(__aarch64__) || defined(_M_ARM64)
+#define SIMDJSON_FASTFLOAT_NEON 1
+#endif
+
+#if defined(SIMDJSON_FASTFLOAT_SSE2) || defined(SIMDJSON_FASTFLOAT_NEON)
+#define SIMDJSON_FASTFLOAT_HAS_SIMD 1
+#endif
+
+#if defined(__GNUC__)
+// disable -Wcast-align=strict (GCC only)
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS \
+ _Pragma("GCC diagnostic push") \
+ _Pragma("GCC diagnostic ignored \"-Wcast-align\"")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+#endif
+
+#if defined(__GNUC__)
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS _Pragma("GCC diagnostic pop")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#define simdjson_fastfloat_really_inline __forceinline
+#else
+#define simdjson_fastfloat_really_inline inline __attribute__((always_inline))
+#endif
+
+// Branch-probability hint marking the rare slow-path branches as cold, so the
+// optimizer keeps the out-of-line slow-path re-parse off the hot path (and does
+// not duplicate the force-inlined hot scanner into the caller, which bloated
+// the hot frame and hurt ILP on some targets). Used at the call site as
+// if simdjson_fastfloat_unlikely(cond) { ... }
+// (the macro supplies the parentheses). It expands to the standard [[unlikely]]
+// attribute when supported, otherwise to __builtin_expect on GCC/Clang, or
+// to a no-op elsewhere (e.g. pre-C++20 MSVC, which has no equivalent hint).
+#ifdef __has_cpp_attribute
+#if __has_cpp_attribute(unlikely) >= 201803L
+// g++-9 hits hits this branch, but then fails to compile
+// [[unlikely]]. This happens only with g++-9.
+#if !defined(__GNUC__) || (__GNUC__ != 9)
+#define SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#endif
+#endif
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#define simdjson_fastfloat_unlikely(x) (x) [[unlikely]]
+#elif defined(__GNUC__) || defined(__clang__)
+#define simdjson_fastfloat_unlikely(x) (__builtin_expect(!!(x), 0))
+#else
+#define simdjson_fastfloat_unlikely(x) (x)
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_ASSERT
+#define SIMDJSON_FASTFLOAT_ASSERT(x) \
+ { \
+ static_cast<void>(x); \
+ }
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DEBUG_ASSERT
+#define SIMDJSON_FASTFLOAT_DEBUG_ASSERT(x) \
+ { \
+ static_cast<void>(x); \
+ }
+#endif
+
+// rust style `try!()` macro, or `?` operator
+#define SIMDJSON_FASTFLOAT_TRY(x) \
+ { \
+ if (!(x)) \
+ return false; \
+ }
+
+#define SIMDJSON_FASTFLOAT_ENABLE_IF(...) \
+ typename std::enable_if<(__VA_ARGS__), int>::type
+
+namespace simdjson_fast_float {
+
+simdjson_fastfloat_really_inline constexpr bool cpp20_and_in_constexpr() {
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED
+ return std::is_constant_evaluated();
+#else
+ return false;
+#endif
+}
+
+template <typename T>
+struct is_supported_float_type
+ : std::integral_constant<
+ bool, std::is_same<T, double>::value || std::is_same<T, float>::value
+#ifdef __STDCPP_FLOAT64_T__
+ || std::is_same<T, std::float64_t>::value
+#endif
+#ifdef __STDCPP_FLOAT32_T__
+ || std::is_same<T, std::float32_t>::value
+#endif
+#ifdef __STDCPP_FLOAT16_T__
+ || std::is_same<T, std::float16_t>::value
+#endif
+#ifdef __STDCPP_BFLOAT16_T__
+ || std::is_same<T, std::bfloat16_t>::value
+#endif
+ > {
+};
+
+template <typename T>
+using equiv_uint_t = typename std::conditional<
+ sizeof(T) == 1, uint8_t,
+ typename std::conditional<
+ sizeof(T) == 2, uint16_t,
+ typename std::conditional<sizeof(T) == 4, uint32_t,
+ uint64_t>::type>::type>::type;
+
+template <typename T> struct is_supported_integer_type : std::is_integral<T> {};
+
+template <typename UC>
+struct is_supported_char_type
+ : std::integral_constant<bool, std::is_same<UC, char>::value ||
+ std::is_same<UC, wchar_t>::value ||
+ std::is_same<UC, char16_t>::value ||
+ std::is_same<UC, char32_t>::value
+#ifdef __cpp_char8_t
+ || std::is_same<UC, char8_t>::value
+#endif
+ > {
+};
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp3(UC const *actual_mixedcase,
+ UC const *expected_lowercase) {
+ uint64_t mask{0};
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ }
+ else {
+ return false;
+ }
+
+ uint64_t val1{0}, val2{0};
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < 3; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1 || sizeof(UC) == 2) {
+ ::memcpy(&val1, actual_mixedcase, 3 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 3 * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ return val1 == val2;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[2] | 32) == (expected_lowercase[2]);
+ }
+ else {
+ return false;
+ }
+ }
+}
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp5(UC const *actual_mixedcase,
+ UC const *expected_lowercase) {
+ uint64_t mask{0};
+ uint64_t val1{0}, val2{0};
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < 5; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) {
+ mask = 0x2020202020202020;
+ ::memcpy(&val1, actual_mixedcase, 5 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 5 * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ return val1 == val2;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ ::memcpy(&val1, actual_mixedcase, 4 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 4 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ ::memcpy(&val1, actual_mixedcase + 2, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase + 2, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+ }
+ else {
+ return false;
+ }
+ }
+}
+
+// Compares two ASCII strings in a case insensitive manner.
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp(UC const *actual_mixedcase, UC const *expected_lowercase,
+ size_t length) {
+ uint64_t mask{0};
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ }
+ else {
+ return false;
+ }
+
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < length; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ uint64_t val1{0}, val2{0};
+ size_t sz{8 / (sizeof(UC))};
+ for (size_t i = 0; i < length; i += sz) {
+ val1 = val2 = 0;
+ sz = sz < (length - i) ? sz : length - i;
+ ::memcpy(&val1, actual_mixedcase + i, sz * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase + i, sz * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ }
+ return true;
+ }
+}
+
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+
+// a pointer and a length to a contiguous block of memory
+template <typename T> struct span {
+ T const *ptr;
+ size_t length;
+
+ constexpr span(T const *_ptr, size_t _length) : ptr(_ptr), length(_length) {}
+
+ constexpr span() : ptr(nullptr), length(0) {}
+
+ constexpr size_t len() const noexcept { return length; }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const T &operator[](size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return ptr[index];
+ }
+};
+
+struct value128 {
+ uint64_t low;
+ uint64_t high;
+
+ constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
+
+ constexpr value128() : low(0), high(0) {}
+};
+
+/* Helper C++14 constexpr generic implementation of leading_zeroes */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+leading_zeroes_generic(uint64_t input_num, int last_bit = 0) {
+ if (input_num & uint64_t(0xffffffff00000000)) {
+ input_num >>= 32;
+ last_bit |= 32;
+ }
+ if (input_num & uint64_t(0xffff0000)) {
+ input_num >>= 16;
+ last_bit |= 16;
+ }
+ if (input_num & uint64_t(0xff00)) {
+ input_num >>= 8;
+ last_bit |= 8;
+ }
+ if (input_num & uint64_t(0xf0)) {
+ input_num >>= 4;
+ last_bit |= 4;
+ }
+ if (input_num & uint64_t(0xc)) {
+ input_num >>= 2;
+ last_bit |= 2;
+ }
+ if (input_num & uint64_t(0x2)) { /* input_num >>= 1; */
+ last_bit |= 1;
+ }
+ return 63 - last_bit;
+}
+
+/* result might be undefined when input_num is zero */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+leading_zeroes(uint64_t input_num) {
+ assert(input_num > 0);
+ if (cpp20_and_in_constexpr()) {
+ return leading_zeroes_generic(input_num);
+ }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#if defined(_M_X64) || defined(_M_ARM64)
+ unsigned long leading_zero = 0;
+ // Search the mask data from most significant bit (MSB)
+ // to least significant bit (LSB) for a set bit (1).
+ _BitScanReverse64(&leading_zero, input_num);
+ return static_cast<int>(63 - leading_zero);
+#else
+ return leading_zeroes_generic(input_num);
+#endif
+#else
+ return __builtin_clzll(input_num);
+#endif
+}
+
+/* Helper C++14 constexpr generic implementation of countr_zero for 32-bit */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+countr_zero_generic_32(uint32_t input_num) {
+ if (input_num == 0) {
+ return 32;
+ }
+ int last_bit = 0;
+ if (!(input_num & 0x0000FFFF)) {
+ input_num >>= 16;
+ last_bit |= 16;
+ }
+ if (!(input_num & 0x00FF)) {
+ input_num >>= 8;
+ last_bit |= 8;
+ }
+ if (!(input_num & 0x0F)) {
+ input_num >>= 4;
+ last_bit |= 4;
+ }
+ if (!(input_num & 0x3)) {
+ input_num >>= 2;
+ last_bit |= 2;
+ }
+ if (!(input_num & 0x1)) {
+ last_bit |= 1;
+ }
+ return last_bit;
+}
+
+/* count trailing zeroes for 32-bit integers */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+countr_zero_32(uint32_t input_num) {
+ if (cpp20_and_in_constexpr()) {
+ return countr_zero_generic_32(input_num);
+ }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+ unsigned long trailing_zero = 0;
+ if (_BitScanForward(&trailing_zero, input_num)) {
+ return static_cast<int>(trailing_zero);
+ }
+ return 32;
+#else
+ return input_num == 0 ? 32 : __builtin_ctz(input_num);
+#endif
+}
+
+// slow emulation routine for 32-bit
+simdjson_fastfloat_really_inline constexpr uint64_t emulu(uint32_t x, uint32_t y) {
+ return x * static_cast<uint64_t>(y);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
+ uint64_t ad =
+ emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd));
+ uint64_t bd = emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd));
+ uint64_t adbc =
+ ad + emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd >> 32));
+ uint64_t adbc_carry = static_cast<uint64_t>(adbc < ad);
+ uint64_t lo = bd + (adbc << 32);
+ *hi =
+ emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd >> 32)) +
+ (adbc >> 32) + (adbc_carry << 32) + static_cast<uint64_t>(lo < bd);
+ return lo;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_32BIT
+
+// slow emulation routine for 32-bit
+#if !defined(__MINGW64__)
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t _umul128(uint64_t ab,
+ uint64_t cd,
+ uint64_t *hi) {
+ return umul128_generic(ab, cd, hi);
+}
+#endif // !__MINGW64__
+
+#endif // SIMDJSON_FASTFLOAT_32BIT
+
+// compute 64-bit a*b
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+full_multiplication(uint64_t a, uint64_t b) {
+ if (cpp20_and_in_constexpr()) {
+ value128 answer;
+ answer.low = umul128_generic(a, b, &answer.high);
+ return answer;
+ }
+ value128 answer;
+#if defined(_M_ARM64) && !defined(__MINGW32__)
+ // ARM64 has native support for 64-bit multiplications, no need to emulate
+ // But MinGW on ARM64 doesn't have native support for 64-bit multiplications
+ answer.high = __umulh(a, b);
+ answer.low = a * b;
+#elif defined(SIMDJSON_FASTFLOAT_32BIT) || (defined(_WIN64) && !defined(__clang__) && \
+ !defined(_M_ARM64) && !defined(__GNUC__))
+ answer.low = _umul128(a, b, &answer.high); // _umul128 not available on ARM64
+#elif defined(SIMDJSON_FASTFLOAT_64BIT) && defined(__SIZEOF_INT128__)
+ __uint128_t r = static_cast<__uint128_t>(a) * b;
+ answer.low = uint64_t(r);
+ answer.high = uint64_t(r >> 64);
+#else
+ answer.low = umul128_generic(a, b, &answer.high);
+#endif
+ return answer;
+}
+
+struct adjusted_mantissa {
+ uint64_t mantissa{0};
+ int32_t power2{0}; // a negative value indicates an invalid result
+ adjusted_mantissa() = default;
+
+ constexpr bool operator==(adjusted_mantissa const &o) const {
+ return mantissa == o.mantissa && power2 == o.power2;
+ }
+
+ constexpr bool operator!=(adjusted_mantissa const &o) const {
+ return mantissa != o.mantissa || power2 != o.power2;
+ }
+};
+
+// Bias so we can get the real exponent with an invalid adjusted_mantissa.
+constexpr static int32_t invalid_am_bias = -0x8000;
+
+// used for binary_format_lookup_tables<T>::max_mantissa
+constexpr uint64_t constant_55555 = 5 * 5 * 5 * 5 * 5;
+
+template <typename T, typename U = void> struct binary_format_lookup_tables;
+
+template <typename T> struct binary_format : binary_format_lookup_tables<T> {
+ using equiv_uint = equiv_uint_t<T>;
+
+ static constexpr int mantissa_explicit_bits();
+ static constexpr int minimum_exponent();
+ static constexpr int infinite_power();
+ static constexpr int sign_index();
+ static constexpr int
+ min_exponent_fast_path(); // used when fegetround() == FE_TONEAREST
+ static constexpr int max_exponent_fast_path();
+ static constexpr int max_exponent_round_to_even();
+ static constexpr int min_exponent_round_to_even();
+ static constexpr uint64_t max_mantissa_fast_path(int64_t power);
+ static constexpr uint64_t
+ max_mantissa_fast_path(); // used when fegetround() == FE_TONEAREST
+ static constexpr int largest_power_of_ten();
+ static constexpr int smallest_power_of_ten();
+ static constexpr T exact_power_of_ten(int64_t power);
+ static constexpr size_t max_digits();
+ static constexpr equiv_uint exponent_mask();
+ static constexpr equiv_uint mantissa_mask();
+ static constexpr equiv_uint hidden_bit_mask();
+};
+
+template <typename U> struct binary_format_lookup_tables<double, U> {
+ static constexpr double powers_of_ten[] = {
+ 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
+ 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22};
+
+ // Largest integer value v so that (5**index * v) <= 1<<53.
+ // 0x20000000000000 == 1 << 53
+ static constexpr uint64_t max_mantissa[] = {
+ 0x20000000000000,
+ 0x20000000000000 / 5,
+ 0x20000000000000 / (5 * 5),
+ 0x20000000000000 / (5 * 5 * 5),
+ 0x20000000000000 / (5 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555),
+ 0x20000000000000 / (constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr double binary_format_lookup_tables<double, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<double, U>::max_mantissa[];
+
+#endif
+
+template <typename U> struct binary_format_lookup_tables<float, U> {
+ static constexpr float powers_of_ten[] = {1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+
+ // Largest integer value v so that (5**index * v) <= 1<<24.
+ // 0x1000000 == 1<<24
+ static constexpr uint64_t max_mantissa[] = {
+ 0x1000000,
+ 0x1000000 / 5,
+ 0x1000000 / (5 * 5),
+ 0x1000000 / (5 * 5 * 5),
+ 0x1000000 / (5 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555),
+ 0x1000000 / (constant_55555 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555 * constant_55555),
+ 0x1000000 / (constant_55555 * constant_55555 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr float binary_format_lookup_tables<float, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<float, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return 0;
+#else
+ return -22;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return 0;
+#else
+ return -10;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<double>::mantissa_explicit_bits() {
+ return 52;
+}
+
+template <>
+inline constexpr int binary_format<float>::mantissa_explicit_bits() {
+ return 23;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_round_to_even() {
+ return 23;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_round_to_even() {
+ return 10;
+}
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_round_to_even() {
+ return -4;
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_round_to_even() {
+ return -17;
+}
+
+template <> inline constexpr int binary_format<double>::minimum_exponent() {
+ return -1023;
+}
+
+template <> inline constexpr int binary_format<float>::minimum_exponent() {
+ return -127;
+}
+
+template <> inline constexpr int binary_format<double>::infinite_power() {
+ return 0x7FF;
+}
+
+template <> inline constexpr int binary_format<float>::infinite_power() {
+ return 0xFF;
+}
+
+template <> inline constexpr int binary_format<double>::sign_index() {
+ return 63;
+}
+
+template <> inline constexpr int binary_format<float>::sign_index() {
+ return 31;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_fast_path() {
+ return 22;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_fast_path() {
+ return 10;
+}
+
+template <>
+inline constexpr uint64_t binary_format<double>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t binary_format<float>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_FLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::float16_t, U> {
+ static constexpr std::float16_t powers_of_ten[] = {1e0f16, 1e1f16, 1e2f16,
+ 1e3f16, 1e4f16};
+
+ // Largest integer value v so that (5**index * v) <= 1<<11.
+ // 0x800 == 1<<11
+ static constexpr uint64_t max_mantissa[] = {0x800,
+ 0x800 / 5,
+ 0x800 / (5 * 5),
+ 0x800 / (5 * 5 * 5),
+ 0x800 / (5 * 5 * 5 * 5),
+ 0x800 / (constant_55555)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::float16_t
+ binary_format_lookup_tables<std::float16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+ binary_format_lookup_tables<std::float16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::float16_t
+binary_format<std::float16_t>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::exponent_mask() {
+ return 0x7C00;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::mantissa_mask() {
+ return 0x03FF;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::hidden_bit_mask() {
+ return 0x0400;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::max_exponent_fast_path() {
+ return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::mantissa_explicit_bits() {
+ return 10;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 4
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::min_exponent_fast_path() {
+ return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::max_exponent_round_to_even() {
+ return 5;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::min_exponent_round_to_even() {
+ return -22;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::minimum_exponent() {
+ return -15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::infinite_power() {
+ return 0x1F;
+}
+
+template <> inline constexpr int binary_format<std::float16_t>::sign_index() {
+ return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::largest_power_of_ten() {
+ return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::smallest_power_of_ten() {
+ return -27;
+}
+
+template <>
+inline constexpr size_t binary_format<std::float16_t>::max_digits() {
+ return 22;
+}
+#endif // __STDCPP_FLOAT16_T__
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_BFLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::bfloat16_t, U> {
+ static constexpr std::bfloat16_t powers_of_ten[] = {1e0bf16, 1e1bf16, 1e2bf16,
+ 1e3bf16};
+
+ // Largest integer value v so that (5**index * v) <= 1<<8.
+ // 0x100 == 1<<8
+ static constexpr uint64_t max_mantissa[] = {0x100, 0x100 / 5, 0x100 / (5 * 5),
+ 0x100 / (5 * 5 * 5),
+ 0x100 / (5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::bfloat16_t
+ binary_format_lookup_tables<std::bfloat16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+ binary_format_lookup_tables<std::bfloat16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::bfloat16_t
+binary_format<std::bfloat16_t>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::max_exponent_fast_path() {
+ return 3;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::exponent_mask() {
+ return 0x7F80;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::mantissa_mask() {
+ return 0x007F;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::hidden_bit_mask() {
+ return 0x0080;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::mantissa_explicit_bits() {
+ return 7;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 3
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::min_exponent_fast_path() {
+ return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::max_exponent_round_to_even() {
+ return 3;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::min_exponent_round_to_even() {
+ return -24;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::minimum_exponent() {
+ return -127;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::infinite_power() {
+ return 0xFF;
+}
+
+template <> inline constexpr int binary_format<std::bfloat16_t>::sign_index() {
+ return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::largest_power_of_ten() {
+ return 38;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::smallest_power_of_ten() {
+ return -60;
+}
+
+template <>
+inline constexpr size_t binary_format<std::bfloat16_t>::max_digits() {
+ return 98;
+}
+#endif // __STDCPP_BFLOAT16_T__
+
+template <>
+inline constexpr uint64_t
+binary_format<double>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 22
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<float>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 10
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr double
+binary_format<double>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr float binary_format<float>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <> inline constexpr int binary_format<double>::largest_power_of_ten() {
+ return 308;
+}
+
+template <> inline constexpr int binary_format<float>::largest_power_of_ten() {
+ return 38;
+}
+
+template <>
+inline constexpr int binary_format<double>::smallest_power_of_ten() {
+ return -342;
+}
+
+template <> inline constexpr int binary_format<float>::smallest_power_of_ten() {
+ return -64;
+}
+
+template <> inline constexpr size_t binary_format<double>::max_digits() {
+ return 769;
+}
+
+template <> inline constexpr size_t binary_format<float>::max_digits() {
+ return 114;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::exponent_mask() {
+ return 0x7F800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::exponent_mask() {
+ return 0x7FF0000000000000;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::mantissa_mask() {
+ return 0x007FFFFF;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::mantissa_mask() {
+ return 0x000FFFFFFFFFFFFF;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::hidden_bit_mask() {
+ return 0x00800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::hidden_bit_mask() {
+ return 0x0010000000000000;
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+to_float(bool negative, adjusted_mantissa am, T &value) {
+ using equiv_uint = equiv_uint_t<T>;
+ equiv_uint word = equiv_uint(am.mantissa);
+ word = equiv_uint(word | equiv_uint(am.power2)
+ << binary_format<T>::mantissa_explicit_bits());
+ word =
+ equiv_uint(word | equiv_uint(negative) << binary_format<T>::sign_index());
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ value = std::bit_cast<T>(word);
+#else
+ ::memcpy(&value, &word, sizeof(T));
+#endif
+}
+
+template <typename = void> struct space_lut {
+ static constexpr bool value[] = {
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr bool space_lut<T>::value[];
+
+#endif
+
+template <typename UC> constexpr bool is_space(UC c) {
+ // wchar_t and char can be signed, so a negative code unit slips past a plain
+ // `c < 256` and then indexes the table by its truncated low byte. Compare as
+ // unsigned, matching the care taken in ch_to_digit.
+ using UnsignedUC = typename std::make_unsigned<UC>::type;
+ return static_cast<UnsignedUC>(c) < 256 && space_lut<>::value[uint8_t(c)];
+}
+
+template <typename UC> static constexpr uint64_t int_cmp_zeros() {
+ static_assert((sizeof(UC) == 1) || (sizeof(UC) == 2) || (sizeof(UC) == 4),
+ "Unsupported character size");
+ return (sizeof(UC) == 1) ? 0x3030303030303030
+ : (sizeof(UC) == 2)
+ ? (uint64_t(UC('0')) << 48 | uint64_t(UC('0')) << 32 |
+ uint64_t(UC('0')) << 16 | UC('0'))
+ : (uint64_t(UC('0')) << 32 | UC('0'));
+}
+
+template <typename UC> static constexpr int int_cmp_len() {
+ return sizeof(uint64_t) / sizeof(UC);
+}
+
+template <typename UC> constexpr UC const *str_const_nan();
+
+template <> constexpr char const *str_const_nan<char>() { return "nan"; }
+
+template <> constexpr wchar_t const *str_const_nan<wchar_t>() { return L"nan"; }
+
+template <> constexpr char16_t const *str_const_nan<char16_t>() {
+ return u"nan";
+}
+
+template <> constexpr char32_t const *str_const_nan<char32_t>() {
+ return U"nan";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_nan<char8_t>() {
+ return u8"nan";
+}
+#endif
+
+template <typename UC> constexpr UC const *str_const_inf();
+
+template <> constexpr char const *str_const_inf<char>() { return "infinity"; }
+
+template <> constexpr wchar_t const *str_const_inf<wchar_t>() {
+ return L"infinity";
+}
+
+template <> constexpr char16_t const *str_const_inf<char16_t>() {
+ return u"infinity";
+}
+
+template <> constexpr char32_t const *str_const_inf<char32_t>() {
+ return U"infinity";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_inf<char8_t>() {
+ return u8"infinity";
+}
+#endif
+
+template <typename = void> struct int_luts {
+ static constexpr uint8_t chdigit[] = {
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 255, 255,
+ 255, 255, 255, 255, 255, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19,
+ 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34,
+ 35, 255, 255, 255, 255, 255, 255, 10, 11, 12, 13, 14, 15, 16, 17,
+ 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32,
+ 33, 34, 35, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255};
+
+ static constexpr size_t maxdigits_u64[] = {
+ 64, 41, 32, 28, 25, 23, 22, 21, 20, 19, 18, 18, 17, 17, 16, 16, 16, 16,
+ 15, 15, 15, 15, 14, 14, 14, 14, 14, 14, 14, 13, 13, 13, 13, 13, 13};
+
+ static constexpr uint64_t min_safe_u64[] = {
+ 9223372036854775808ull, 12157665459056928801ull, 4611686018427387904,
+ 7450580596923828125, 4738381338321616896, 3909821048582988049,
+ 9223372036854775808ull, 12157665459056928801ull, 10000000000000000000ull,
+ 5559917313492231481, 2218611106740436992, 8650415919381337933,
+ 2177953337809371136, 6568408355712890625, 1152921504606846976,
+ 2862423051509815793, 6746640616477458432, 15181127029874798299ull,
+ 1638400000000000000, 3243919932521508681, 6221821273427820544,
+ 11592836324538749809ull, 876488338465357824, 1490116119384765625,
+ 2481152873203736576, 4052555153018976267, 6502111422497947648,
+ 10260628712958602189ull, 15943230000000000000ull, 787662783788549761,
+ 1152921504606846976, 1667889514952984961, 2386420683693101056,
+ 3379220508056640625, 4738381338321616896};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint8_t int_luts<T>::chdigit[];
+
+template <typename T> constexpr size_t int_luts<T>::maxdigits_u64[];
+
+template <typename T> constexpr uint64_t int_luts<T>::min_safe_u64[];
+
+#endif
+
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr uint8_t ch_to_digit(UC c) {
+ // wchar_t and char can be signed, so we need to be careful.
+ using UnsignedUC = typename std::make_unsigned<UC>::type;
+ return int_luts<>::chdigit[static_cast<unsigned char>(
+ static_cast<UnsignedUC>(c) &
+ static_cast<UnsignedUC>(
+ -((static_cast<UnsignedUC>(c) & ~0xFFull) == 0)))];
+}
+
+simdjson_fastfloat_really_inline constexpr size_t max_digits_u64(int base) {
+ return int_luts<>::maxdigits_u64[base - 2];
+}
+
+// If a u64 is exactly max_digits_u64() in length, this is
+// the value below which it has definitely overflowed.
+simdjson_fastfloat_really_inline constexpr uint64_t min_safe_u64(int base) {
+ return int_luts<>::min_safe_u64[base - 2];
+}
+
+static_assert(std::is_same<equiv_uint_t<double>, uint64_t>::value,
+ "equiv_uint should be uint64_t for double");
+static_assert(std::numeric_limits<double>::is_iec559,
+ "double must fulfill the requirements of IEC 559 (IEEE 754)");
+
+static_assert(std::is_same<equiv_uint_t<float>, uint32_t>::value,
+ "equiv_uint should be uint32_t for float");
+static_assert(std::numeric_limits<float>::is_iec559,
+ "float must fulfill the requirements of IEC 559 (IEEE 754)");
+
+#ifdef __STDCPP_FLOAT64_T__
+static_assert(std::is_same<equiv_uint_t<std::float64_t>, uint64_t>::value,
+ "equiv_uint should be uint64_t for std::float64_t");
+static_assert(
+ std::numeric_limits<std::float64_t>::is_iec559,
+ "std::float64_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float64_t> : public binary_format<double> {};
+#endif // __STDCPP_FLOAT64_T__
+
+#ifdef __STDCPP_FLOAT32_T__
+static_assert(std::is_same<equiv_uint_t<std::float32_t>, uint32_t>::value,
+ "equiv_uint should be uint32_t for std::float32_t");
+static_assert(
+ std::numeric_limits<std::float32_t>::is_iec559,
+ "std::float32_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float32_t> : public binary_format<float> {};
+#endif // __STDCPP_FLOAT32_T__
+
+#ifdef __STDCPP_FLOAT16_T__
+static_assert(
+ std::is_same<binary_format<std::float16_t>::equiv_uint, uint16_t>::value,
+ "equiv_uint should be uint16_t for std::float16_t");
+static_assert(
+ std::numeric_limits<std::float16_t>::is_iec559,
+ "std::float16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_FLOAT16_T__
+
+#ifdef __STDCPP_BFLOAT16_T__
+static_assert(
+ std::is_same<binary_format<std::bfloat16_t>::equiv_uint, uint16_t>::value,
+ "equiv_uint should be uint16_t for std::bfloat16_t");
+static_assert(
+ std::numeric_limits<std::bfloat16_t>::is_iec559,
+ "std::bfloat16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_BFLOAT16_T__
+
+constexpr chars_format operator~(chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(~static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator&(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) &
+ static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator|(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) |
+ static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator^(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) ^
+ static_cast<int_type>(rhs));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator&=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs & rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator|=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs | rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator^=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs ^ rhs);
+}
+
+namespace detail {
+// adjust for deprecated feature macros
+constexpr chars_format adjust_for_feature_macros(chars_format fmt) {
+ return fmt
+#ifdef SIMDJSON_FASTFLOAT_ALLOWS_LEADING_PLUS
+ | chars_format::allow_leading_plus
+#endif
+#ifdef SIMDJSON_FASTFLOAT_SKIP_WHITE_SPACE
+ | chars_format::skip_white_space
+#endif
+ ;
+}
+} // namespace detail
+} // namespace simdjson_fast_float
+
+#endif
+
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+#define SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+
+namespace simdjson_fast_float {
+/**
+ * This function parses the character sequence [first,last) for a number. It
+ * parses floating-point numbers expecting a locale-independent format
+ * equivalent to what is used by std::strtod in the default ("C") locale. The
+ * resulting floating-point value is the closest floating-point values (using
+ * either float or double), using the "round to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * parsing according to the IEEE standard.
+ *
+ * Given a successful parse, the pointer (`ptr`) in the returned value is set to
+ * point right after the parsed number, and the `value` referenced is set to the
+ * parsed value. In case of error, the returned `ec` contains a representative
+ * error, otherwise the default (`std::errc()`) value is stored.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ *
+ * Like the C++17 standard, the `simdjson_fast_float::from_chars` functions take an
+ * optional last argument of the type `simdjson_fast_float::chars_format`. It is a bitset
+ * value: we check whether `fmt & simdjson_fast_float::chars_format::fixed` and `fmt &
+ * simdjson_fast_float::chars_format::scientific` are set to determine whether we allow
+ * the fixed point and scientific notation respectively. The default is
+ * `simdjson_fast_float::chars_format::general` which allows both `fixed` and
+ * `scientific`.
+ */
+template <typename T, typename UC = char,
+ typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_float_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+ chars_format fmt = chars_format::general) noexcept;
+
+/**
+ * Like from_chars, but accepts an `options` argument to govern number parsing.
+ * Both for floating-point types and integer types.
+ */
+template <typename T, typename UC = char>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept;
+
+/**
+ * This function multiplies an integer number by a power of 10 and returns
+ * the result as a double precision floating-point value that is correctly
+ * rounded. The resulting floating-point value is the closest floating-point
+ * value, using the "round to nearest, tie to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * conversion according to the IEEE standard.
+ *
+ * On overflow infinity is returned, on underflow 0 is returned.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ */
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * This function is a template overload of `integer_times_pow10()`
+ * that returns a floating-point value of type `T` that is one of
+ * supported floating-point types (e.g. `double`, `float`).
+ */
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * from_chars for integer types.
+ */
+template <typename T, typename UC = char,
+ typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_integer_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base = 10) noexcept;
+
+} // namespace simdjson_fast_float
+
+#endif // SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+#ifndef SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+#define SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+
+#include <cctype>
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+#include <limits>
+#include <type_traits>
+
+
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+#include <emmintrin.h>
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_NEON
+#include <arm_neon.h>
+#endif
+
+namespace simdjson_fast_float {
+
+template <typename UC> simdjson_fastfloat_really_inline constexpr bool has_simd_opt() {
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+ return std::is_same<UC, char16_t>::value;
+#else
+ return false;
+#endif
+}
+
+// Next function can be micro-optimized, but compilers are entirely
+// able to optimize it well.
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr bool is_integer(UC c) noexcept {
+ return static_cast<unsigned>(c - UC('0')) <= 9u;
+}
+
+simdjson_fastfloat_really_inline constexpr uint64_t byteswap(uint64_t val) {
+ return (val & 0xFF00000000000000) >> 56 | (val & 0x00FF000000000000) >> 40 |
+ (val & 0x0000FF0000000000) >> 24 | (val & 0x000000FF00000000) >> 8 |
+ (val & 0x00000000FF000000) << 8 | (val & 0x0000000000FF0000) << 24 |
+ (val & 0x000000000000FF00) << 40 | (val & 0x00000000000000FF) << 56;
+}
+
+simdjson_fastfloat_really_inline constexpr uint32_t byteswap_32(uint32_t val) {
+ return (val >> 24) | ((val >> 8) & 0x0000FF00u) | ((val << 8) & 0x00FF0000u) |
+ (val << 24);
+}
+
+// Read 8 UC into a u64. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+read8_to_u64(UC const *chars) {
+ if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+ uint64_t val = 0;
+ for (int i = 0; i < 8; ++i) {
+ val |= uint64_t(uint8_t(*chars)) << (i * 8);
+ ++chars;
+ }
+ return val;
+ }
+ uint64_t val;
+ ::memcpy(&val, chars, sizeof(uint64_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+ // Need to read as-if the number was in little-endian order.
+ val = byteswap(val);
+#endif
+ return val;
+}
+
+// Read 4 UC into a u32. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+read4_to_u32(UC const *chars) {
+ if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+ uint32_t val = 0;
+ for (int i = 0; i < 4; ++i) {
+ val |= uint32_t(uint8_t(*chars)) << (i * 8);
+ ++chars;
+ }
+ return val;
+ }
+ uint32_t val;
+ ::memcpy(&val, chars, sizeof(uint32_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+ val = byteswap_32(val);
+#endif
+ return val;
+}
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(__m128i const data) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ __m128i const packed = _mm_packus_epi16(data, data);
+#ifdef SIMDJSON_FASTFLOAT_64BIT
+ return uint64_t(_mm_cvtsi128_si64(packed));
+#else
+ uint64_t value;
+ // Visual Studio + older versions of GCC don't support _mm_storeu_si64
+ _mm_storel_epi64(reinterpret_cast<__m128i *>(&value), packed);
+ return value;
+#endif
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ return simd_read8_to_u64(
+ _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars)));
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(uint16x8_t const data) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ uint8x8_t utf8_packed = vmovn_u16(data);
+ return vget_lane_u64(vreinterpret_u64_u8(utf8_packed), 0);
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ return simd_read8_to_u64(
+ vld1q_u16(reinterpret_cast<uint16_t const *>(chars)));
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#endif // SIMDJSON_FASTFLOAT_SSE2
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+uint64_t simd_read8_to_u64(UC const *) {
+ return 0;
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_eight_digits_unrolled(uint64_t val) {
+ uint64_t const mask = 0x000000FF000000FF;
+ uint64_t const mul1 = 0x000F424000000064; // 100 + (1000000ULL << 32)
+ uint64_t const mul2 = 0x0000271000000001; // 1 + (10000ULL << 32)
+ val -= 0x3030303030303030;
+ val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
+ val = (((val & mask) * mul1) + (((val >> 16) & mask) * mul2)) >> 32;
+ return uint32_t(val);
+}
+
+// Call this if chars are definitely 8 digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+parse_eight_digits_unrolled(UC const *chars) noexcept {
+ if (cpp20_and_in_constexpr() || !has_simd_opt<UC>()) {
+ return parse_eight_digits_unrolled(read8_to_u64(chars)); // truncation okay
+ }
+ return parse_eight_digits_unrolled(simd_read8_to_u64(chars));
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_eight_digits_fast(uint64_t val) noexcept {
+ return !((((val + 0x4646464646464646) | (val - 0x3030303030303030)) &
+ 0x8080808080808080));
+}
+
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_four_digits_fast(uint32_t val) noexcept {
+ return !((((val + 0x46464646) | (val - 0x30303030)) & 0x80808080));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_four_digits_unrolled(uint32_t val) noexcept {
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// Call this if chars might not be 8 digits.
+// Using this style (instead of is_made_of_eight_digits_fast() then
+// parse_eight_digits_unrolled()) ensures we don't load SIMD registers twice.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+simd_parse_if_eight_digits_unrolled(char16_t const *chars,
+ uint64_t &i) noexcept {
+ if (cpp20_and_in_constexpr()) {
+ return false;
+ }
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ __m128i const data =
+ _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars));
+
+ // (x - '0') <= 9
+ // http://0x80.pl/articles/simd-parsing-int-sequences.html
+ __m128i const t0 = _mm_add_epi16(data, _mm_set1_epi16(32720));
+ __m128i const t1 = _mm_cmpgt_epi16(t0, _mm_set1_epi16(-32759));
+
+ if (_mm_movemask_epi8(t1) == 0) {
+ i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+ return true;
+ } else
+ return false;
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ uint16x8_t const data = vld1q_u16(reinterpret_cast<uint16_t const *>(chars));
+
+ // (x - '0') <= 9
+ // http://0x80.pl/articles/simd-parsing-int-sequences.html
+ uint16x8_t const t0 = vsubq_u16(data, vmovq_n_u16('0'));
+ uint16x8_t const mask = vcltq_u16(t0, vmovq_n_u16('9' - '0' + 1));
+
+ if (vminvq_u16(mask) == 0xFFFF) {
+ i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+ return true;
+ } else
+ return false;
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#else
+ static_cast<void>(chars);
+ static_cast<void>(i);
+ return false;
+#endif // SIMDJSON_FASTFLOAT_SSE2
+}
+
+#endif // SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+bool simd_parse_if_eight_digits_unrolled(UC const *, uint64_t &) {
+ return 0;
+}
+
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!std::is_same<UC, char>::value) = 0>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(UC const *&p, UC const *const pend, uint64_t &i) {
+ if (!has_simd_opt<UC>()) {
+ return;
+ }
+ while ((std::distance(p, pend) >= 8) &&
+ simd_parse_if_eight_digits_unrolled(
+ p, i)) { // in rare cases, this will overflow, but that's ok
+ p += 8;
+ }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(char const *&p, char const *const pend,
+ uint64_t &i) {
+ // optimizes better than parse_if_eight_digits_unrolled() for UC = char.
+ while ((std::distance(p, pend) >= 8) &&
+ is_made_of_eight_digits_fast(read8_to_u64(p))) {
+ i = i * 100000000 +
+ parse_eight_digits_unrolled(read8_to_u64(
+ p)); // in rare cases, this will overflow, but that's ok
+ p += 8;
+ }
+ // Consume a remaining 4-7 digit run in a single SWAR step instead of
+ // byte-by-byte (reuses the existing 4-digit helpers). The parsed result is
+ // identical either way. Historically gated to clang because gcc regressed on
+ // short remainders, but that verdict predates the span-elision restructure;
+ // with the leaner hot path the 4-digit step now wins on gcc as well.
+ if ((pend - p) >= 4) {
+ uint32_t const val4 = read4_to_u32(p);
+ if (is_made_of_four_digits_fast(val4)) {
+ i = i * 10000 +
+ parse_four_digits_unrolled(val4); // may overflow, that's ok
+ p += 4;
+ }
+ }
+}
+
+enum class parse_error {
+ no_error,
+ // [JSON-only] The minus sign must be followed by an integer.
+ missing_integer_after_sign,
+ // A sign must be followed by an integer or dot.
+ missing_integer_or_dot_after_sign,
+ // [JSON-only] The integer part must not have leading zeros.
+ leading_zeros_in_integer_part,
+ // [JSON-only] The integer part must have at least one digit.
+ no_digits_in_integer_part,
+ // [JSON-only] If there is a decimal point, there must be digits in the
+ // fractional part.
+ no_digits_in_fractional_part,
+ // The mantissa must have at least one digit.
+ no_digits_in_mantissa,
+ // Scientific notation requires an exponential part.
+ missing_exponential_part,
+};
+
+template <typename UC> struct parsed_number_string_t {
+ int64_t exponent{0};
+ uint64_t mantissa{0};
+ UC const *lastmatch{nullptr};
+ bool negative{false};
+ bool valid{false};
+ bool too_many_digits{false};
+ // contains the range of the significant digits
+ span<UC const> integer{}; // non-nullable
+ span<UC const> fraction{}; // nullable
+ parse_error error{parse_error::no_error};
+};
+
+using byte_span = span<char const>;
+using parsed_number_string = parsed_number_string_t<char>;
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+report_parse_error(UC const *p, parse_error error) {
+ parsed_number_string_t<UC> answer;
+ answer.valid = false;
+ answer.lastmatch = p;
+ answer.error = error;
+ return answer;
+}
+
+// Assuming that you use no more than 19 digits, this will
+// parse an ASCII string.
+//
+// store_spans is a *runtime* flag (not a template parameter, deliberately: a
+// template would create a second instantiation of this whole function and the
+// extra icache pressure wipes out the gain). When false, the integer/fraction
+// spans (read only by the rare digit_comp slow path) are not materialized,
+// which keeps the fat parsed_number_string_t off the hot path. The caller
+// re-parses with store_spans=true if the slow path is actually reached.
+template <bool basic_json_fmt, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+parse_number_string(UC const *p, UC const *pend, parse_options_t<UC> options,
+ bool store_spans = true) noexcept {
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ UC const decimal_point = options.decimal_point;
+
+ parsed_number_string_t<UC> answer;
+ answer.valid = false;
+ answer.too_many_digits = false;
+ // assume p < pend, so dereference without checks;
+ answer.negative = (*p == UC('-'));
+ // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+ if ((*p == UC('-')) || (uint64_t(fmt & chars_format::allow_leading_plus) &&
+ !basic_json_fmt && *p == UC('+'))) {
+ ++p;
+ if (p == pend) {
+ return report_parse_error<UC>(
+ p, parse_error::missing_integer_or_dot_after_sign);
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ if (!is_integer(*p)) { // a sign must be followed by an integer
+ return report_parse_error<UC>(p,
+ parse_error::missing_integer_after_sign);
+ }
+ }
+ else {
+ if (!is_integer(*p) &&
+ (*p !=
+ decimal_point)) { // a sign must be followed by an integer or the dot
+ return report_parse_error<UC>(
+ p, parse_error::missing_integer_or_dot_after_sign);
+ }
+ }
+ }
+ UC const *const start_digits = p;
+
+ uint64_t i = 0; // an unsigned int avoids signed overflows (which are bad)
+
+ // Straight-line unroll of the integer-part scan: most integer parts are
+ // 1-5 digits, so peeling the first iterations eliminates the loop back-edge
+ // for the common case. Semantics are identical to the original `while` loop:
+ // i = 10*i + digit, advancing p.
+ if ((p != pend) && is_integer(*p)) {
+ i = uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ while ((p != pend) && is_integer(*p)) {
+ // a multiplication by 10 is cheaper than an arbitrary integer
+ // multiplication
+ i = 10 * i +
+ uint64_t(*p - UC('0')); // might overflow, handled later
+ ++p;
+ }
+ }
+ }
+ }
+ }
+ }
+ UC const *const end_of_integer_part = p;
+ int64_t digit_count = int64_t(end_of_integer_part - start_digits);
+ if (store_spans) {
+ answer.integer = span<UC const>(start_digits, size_t(digit_count));
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ // at least 1 digit in integer part, without leading zeros
+ if (digit_count == 0) {
+ return report_parse_error<UC>(p, parse_error::no_digits_in_integer_part);
+ }
+ if ((start_digits[0] == UC('0') && digit_count > 1)) {
+ return report_parse_error<UC>(start_digits,
+ parse_error::leading_zeros_in_integer_part);
+ }
+ }
+
+ int64_t exponent = 0;
+ bool const has_decimal_point = (p != pend) && (*p == decimal_point);
+ if (has_decimal_point) {
+ ++p;
+ UC const *before = p;
+ // can occur at most twice without overflowing, but let it occur more, since
+ // for integers with many digits, digit parsing is the primary bottleneck.
+ loop_parse_if_eight_digits(p, pend, i);
+
+ while ((p != pend) && is_integer(*p)) {
+ uint8_t digit = uint8_t(*p - UC('0'));
+ ++p;
+ i = i * 10 + digit; // in rare cases, this will overflow, but that's ok
+ }
+ exponent = before - p;
+ if (store_spans) {
+ answer.fraction = span<UC const>(before, size_t(p - before));
+ }
+ digit_count -= exponent;
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ // at least 1 digit in fractional part
+ if (has_decimal_point && exponent == 0) {
+ return report_parse_error<UC>(p,
+ parse_error::no_digits_in_fractional_part);
+ }
+ }
+ else if (digit_count == 0) { // we must have encountered at least one integer!
+ return report_parse_error<UC>(p, parse_error::no_digits_in_mantissa);
+ }
+ int64_t exp_number = 0; // explicit exponential part
+ if ((uint64_t(fmt & chars_format::scientific) && (p != pend) &&
+ ((UC('e') == *p) || (UC('E') == *p))) ||
+ (uint64_t(fmt & detail::basic_fortran_fmt) && (p != pend) &&
+ ((UC('+') == *p) || (UC('-') == *p) || (UC('d') == *p) ||
+ (UC('D') == *p)))) {
+ UC const *location_of_e = p;
+ if ((UC('e') == *p) || (UC('E') == *p) || (UC('d') == *p) ||
+ (UC('D') == *p)) {
+ ++p;
+ }
+ bool neg_exp = false;
+ if ((p != pend) && (UC('-') == *p)) {
+ neg_exp = true;
+ ++p;
+ } else if ((p != pend) &&
+ (UC('+') ==
+ *p)) { // '+' on exponent is allowed by C++17 20.19.3.(7.1)
+ ++p;
+ }
+ if ((p == pend) || !is_integer(*p)) {
+ if (!uint64_t(fmt & chars_format::fixed)) {
+ // The exponential part is invalid for scientific notation, so it must
+ // be a trailing token for fixed notation. However, fixed notation is
+ // disabled, so report a scientific notation error.
+ return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+ }
+ // Otherwise, we will be ignoring the 'e'.
+ p = location_of_e;
+ } else {
+ while ((p != pend) && is_integer(*p)) {
+ uint8_t digit = uint8_t(*p - UC('0'));
+ if (exp_number < 0x10000000) {
+ exp_number = 10 * exp_number + digit;
+ }
+ ++p;
+ }
+ if (neg_exp) {
+ exp_number = -exp_number;
+ }
+ exponent += exp_number;
+ }
+ } else {
+ // If it scientific and not fixed, we have to bail out.
+ if (uint64_t(fmt & chars_format::scientific) &&
+ !uint64_t(fmt & chars_format::fixed)) {
+ return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+ }
+ }
+ answer.lastmatch = p;
+ answer.valid = true;
+
+ // If we frequently had to deal with long strings of digits,
+ // we could extend our code by using a 128-bit integer instead
+ // of a 64-bit integer. However, this is uncommon.
+ //
+ // We can deal with up to 19 digits.
+ if (digit_count > 19) { // this is uncommon
+ // It is possible that the integer had an overflow.
+ // We have to handle the case where we have 0.0000somenumber.
+ // We need to be mindful of the case where we only have zeroes...
+ // E.g., 0.000000000...000.
+ UC const *start = start_digits;
+ while ((start != pend) && (*start == UC('0') || *start == decimal_point)) {
+ if (*start == UC('0')) {
+ digit_count--;
+ }
+ start++;
+ }
+
+ if (digit_count > 19) {
+ answer.too_many_digits = true;
+ // The truncation recompute below reads the integer/fraction spans. When
+ // store_spans is false we didn't materialize them, so just flag
+ // too_many_digits; the caller re-parses with store_spans=true to obtain
+ // the corrected mantissa/exponent before taking the slow path.
+ if (store_spans) {
+ // Let us start again, this time, avoiding overflows.
+ // We don't need to call if is_integer, since we use the
+ // pre-tokenized spans from above.
+ i = 0;
+ p = answer.integer.ptr;
+ UC const *int_end = p + answer.integer.len();
+ uint64_t const minimal_nineteen_digit_integer{1000000000000000000};
+ while ((i < minimal_nineteen_digit_integer) && (p != int_end)) {
+ i = i * 10 + uint64_t(*p - UC('0'));
+ ++p;
+ }
+ if (i >= minimal_nineteen_digit_integer) { // We have a big integer
+ exponent = end_of_integer_part - p + exp_number;
+ } else { // We have a value with a fractional component.
+ p = answer.fraction.ptr;
+ UC const *frac_end = p + answer.fraction.len();
+ while ((i < minimal_nineteen_digit_integer) && (p != frac_end)) {
+ i = i * 10 + uint64_t(*p - UC('0'));
+ ++p;
+ }
+ exponent = answer.fraction.ptr - p + exp_number;
+ }
+ // We have now corrected both exponent and i, to a truncated value
+ }
+ }
+ }
+ answer.exponent = exponent;
+ answer.mantissa = i;
+ return answer;
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_int_string(UC const *p, UC const *pend, T &value,
+ parse_options_t<UC> options) {
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ int const base = options.base;
+
+ from_chars_result_t<UC> answer;
+
+ UC const *const first = p;
+
+ bool const negative = (*p == UC('-'));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4127)
+#endif
+ if (!std::is_signed<T>::value && negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+ if ((*p == UC('-')) ||
+ (uint64_t(fmt & chars_format::allow_leading_plus) && (*p == UC('+')))) {
+ ++p;
+ }
+
+ UC const *const start_num = p;
+
+ while (p != pend && *p == UC('0')) {
+ ++p;
+ }
+
+ bool const has_leading_zeros = p > start_num;
+
+ UC const *const start_digits = p;
+
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+ (std::is_same<T, std::uint8_t>::value && sizeof(UC) == 1)) {
+ if (base == 10) {
+ const size_t len = static_cast<size_t>(pend - p);
+ if (len == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ uint32_t digits;
+
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ if (std::is_constant_evaluated()) {
+ uint8_t str[4]{};
+ for (size_t j = 0; j < 4 && j < len; ++j) {
+ str[j] = static_cast<uint8_t>(p[j]);
+ }
+ digits = std::bit_cast<uint32_t>(str);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+ digits = byteswap_32(digits);
+#endif
+ }
+#else
+ if (false) {
+ }
+#endif
+ else if (len >= 4) {
+ ::memcpy(&digits, p, 4);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+ digits = byteswap_32(digits);
+#endif
+ } else {
+ uint32_t b0 = static_cast<uint8_t>(p[0]);
+ uint32_t b1 = (len > 1) ? static_cast<uint8_t>(p[1]) : 0xFFu;
+ uint32_t b2 = (len > 2) ? static_cast<uint8_t>(p[2]) : 0xFFu;
+ uint32_t b3 = 0xFFu;
+ digits = b0 | (b1 << 8) | (b2 << 16) | (b3 << 24);
+ }
+
+ uint32_t magic =
+ ((digits + 0x46464646u) | (digits - 0x30303030u)) & 0x80808080u;
+ uint32_t tz =
+ static_cast<uint32_t>(countr_zero_32(magic)); // 7, 15, 23, 31, or 32
+ uint32_t nd = (tz == 32) ? 4 : (tz >> 3);
+ nd = static_cast<uint32_t>(nd < len ? nd : len);
+ if (nd == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ return answer;
+ }
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+ if (nd > 3) {
+ const UC *q = p + nd;
+ size_t rem = len - nd;
+ while (rem) {
+ if (*q < UC('0') || *q > UC('9'))
+ break;
+ ++q;
+ --rem;
+ }
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = q;
+ return answer;
+ }
+
+ digits ^= 0x30303030u;
+ digits <<= ((4 - nd) * 8);
+
+ uint32_t check = ((digits >> 24) & 0xff) | ((digits >> 8) & 0xff00) |
+ ((digits << 8) & 0xff0000);
+ if (check > 0x00020505) {
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = p + nd;
+ return answer;
+ }
+ value = static_cast<uint8_t>((0x640a01 * digits) >> 24);
+ answer.ec = std::errc();
+ answer.ptr = p + nd;
+ return answer;
+ }
+ }
+
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+ (std::is_same<T, std::uint16_t>::value && sizeof(UC) == 1)) {
+ if (base == 10) {
+ const size_t len = size_t(pend - p);
+ if (len == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ if (len >= 4) {
+ uint32_t digits = read4_to_u32(p);
+ if (is_made_of_four_digits_fast(digits)) {
+ uint32_t v = parse_four_digits_unrolled(digits);
+ if (len >= 5 && is_integer(p[4])) {
+ v = v * 10 + uint32_t(p[4] - '0');
+ if (len >= 6 && is_integer(p[5])) {
+ answer.ec = std::errc::result_out_of_range;
+ const UC *q = p + 5;
+ while (q != pend && is_integer(*q)) {
+ q++;
+ }
+ answer.ptr = q;
+ return answer;
+ }
+ if (v > 65535) {
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = p + 5;
+ return answer;
+ }
+ value = uint16_t(v);
+ answer.ec = std::errc();
+ answer.ptr = p + 5;
+ return answer;
+ }
+ // 4 digits
+ value = uint16_t(v);
+ answer.ec = std::errc();
+ answer.ptr = p + 4;
+ return answer;
+ }
+ }
+ }
+ }
+
+ uint64_t i = 0;
+ if (base == 10) {
+ loop_parse_if_eight_digits(p, pend, i); // use SIMD if possible
+ }
+ while (p != pend) {
+ uint8_t digit = ch_to_digit(*p);
+ if (digit >= base) {
+ break;
+ }
+ i = uint64_t(base) * i + digit; // might overflow, check this later
+ p++;
+ }
+
+ size_t digit_count = size_t(p - start_digits);
+
+ if (digit_count == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ answer.ptr = p;
+
+ // check u64 overflow
+ size_t max_digits = max_digits_u64(base);
+ if (digit_count > max_digits) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ // this check can be eliminated for all other types, but they will all require
+ // a max_digits(base) equivalent
+ if (digit_count == max_digits) {
+ // At the max_digits boundary the accumulator `i` may have wrapped around
+ // 2^64. A plain `i < min_safe_u64(base)` test is not sufficient: for any
+ // base whose max_digits-length range exceeds 2^64 (base 10 reaches
+ // ~5.4 * 2^64 at 20 digits) the value can wrap a whole multiple of 2^64 and
+ // land back above min_safe, slipping through. Decide exactly in O(1) using
+ // the leading digit, following the approach used in simdjson:
+ // ms == min_safe_u64(base) == base^(max_digits-1), the smallest
+ // max_digits-length value.
+ // dmax == the largest leading digit whose number can still fit in u64.
+ // The leading-digit band [d*ms, (d+1)*ms) has width ms < 2^64, so within
+ // the single band where d == dmax the value straddles 2^64 at most once,
+ // and a single threshold separates wrapped from non-wrapped values. A
+ // leading digit above dmax always overflows; below dmax always fits.
+ uint64_t const ms = min_safe_u64(base);
+ uint64_t const dmax = (std::numeric_limits<uint64_t>::max)() / ms;
+ uint64_t const lead = ch_to_digit(*start_digits);
+ if (lead > dmax || (lead == dmax && i < dmax * ms)) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ }
+
+ // check other types overflow
+ if (!std::is_same<T, uint64_t>::value) {
+ if (i > uint64_t((std::numeric_limits<T>::max)()) + uint64_t(negative)) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ }
+
+ if (negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4146)
+#endif
+ // this weird workaround is required because:
+ // - converting unsigned to signed when its value is greater than signed max
+ // is UB pre-C++23.
+ // - reinterpret_casting (~i + 1) would work, but it is not constexpr
+ // this is always optimized into a neg instruction (note: T is an integer
+ // type)
+ value = T(-(std::numeric_limits<T>::max)() -
+ T(i - uint64_t((std::numeric_limits<T>::max)())));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+ } else {
+ value = T(i);
+ }
+
+ answer.ec = std::errc();
+ return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_TABLE_H
+#define SIMDJSON_FASTFLOAT_FAST_TABLE_H
+
+#include <cstdint>
+
+namespace simdjson_fast_float {
+
+/**
+ * When mapping numbers from decimal to binary,
+ * we go from w * 10^q to m * 2^p but we have
+ * 10^q = 5^q * 2^q, so effectively
+ * we are trying to match
+ * w * 2^q * 5^q to m * 2^p. Thus the powers of two
+ * are not a concern since they can be represented
+ * exactly using the binary notation, only the powers of five
+ * affect the binary significand.
+ */
+
+/**
+ * The smallest non-zero float (binary64) is 2^-1074.
+ * We take as input numbers of the form w x 10^q where w < 2^64.
+ * We have that w * 10^-343 < 2^(64-344) 5^-343 < 2^-1076.
+ * However, we have that
+ * (2^64-1) * 10^-342 = (2^64-1) * 2^-342 * 5^-342 > 2^-1074.
+ * Thus it is possible for a number of the form w * 10^-342 where
+ * w is a 64-bit value to be a non-zero floating-point number.
+ *********
+ * Any number of form w * 10^309 where w>= 1 is going to be
+ * infinite in binary64 so we never need to worry about powers
+ * of 5 greater than 308.
+ */
+template <class unused = void> struct powers_template {
+
+ constexpr static int smallest_power_of_five =
+ binary_format<double>::smallest_power_of_ten();
+ constexpr static int largest_power_of_five =
+ binary_format<double>::largest_power_of_ten();
+ constexpr static int number_of_entries =
+ 2 * (largest_power_of_five - smallest_power_of_five + 1);
+ // Powers of five from 5^-342 all the way to 5^308 rounded toward one.
+ constexpr static uint64_t power_of_five_128[number_of_entries] = {
+ 0xeef453d6923bd65a, 0x113faa2906a13b3f,
+ 0x9558b4661b6565f8, 0x4ac7ca59a424c507,
+ 0xbaaee17fa23ebf76, 0x5d79bcf00d2df649,
+ 0xe95a99df8ace6f53, 0xf4d82c2c107973dc,
+ 0x91d8a02bb6c10594, 0x79071b9b8a4be869,
+ 0xb64ec836a47146f9, 0x9748e2826cdee284,
+ 0xe3e27a444d8d98b7, 0xfd1b1b2308169b25,
+ 0x8e6d8c6ab0787f72, 0xfe30f0f5e50e20f7,
+ 0xb208ef855c969f4f, 0xbdbd2d335e51a935,
+ 0xde8b2b66b3bc4723, 0xad2c788035e61382,
+ 0x8b16fb203055ac76, 0x4c3bcb5021afcc31,
+ 0xaddcb9e83c6b1793, 0xdf4abe242a1bbf3d,
+ 0xd953e8624b85dd78, 0xd71d6dad34a2af0d,
+ 0x87d4713d6f33aa6b, 0x8672648c40e5ad68,
+ 0xa9c98d8ccb009506, 0x680efdaf511f18c2,
+ 0xd43bf0effdc0ba48, 0x212bd1b2566def2,
+ 0x84a57695fe98746d, 0x14bb630f7604b57,
+ 0xa5ced43b7e3e9188, 0x419ea3bd35385e2d,
+ 0xcf42894a5dce35ea, 0x52064cac828675b9,
+ 0x818995ce7aa0e1b2, 0x7343efebd1940993,
+ 0xa1ebfb4219491a1f, 0x1014ebe6c5f90bf8,
+ 0xca66fa129f9b60a6, 0xd41a26e077774ef6,
+ 0xfd00b897478238d0, 0x8920b098955522b4,
+ 0x9e20735e8cb16382, 0x55b46e5f5d5535b0,
+ 0xc5a890362fddbc62, 0xeb2189f734aa831d,
+ 0xf712b443bbd52b7b, 0xa5e9ec7501d523e4,
+ 0x9a6bb0aa55653b2d, 0x47b233c92125366e,
+ 0xc1069cd4eabe89f8, 0x999ec0bb696e840a,
+ 0xf148440a256e2c76, 0xc00670ea43ca250d,
+ 0x96cd2a865764dbca, 0x380406926a5e5728,
+ 0xbc807527ed3e12bc, 0xc605083704f5ecf2,
+ 0xeba09271e88d976b, 0xf7864a44c633682e,
+ 0x93445b8731587ea3, 0x7ab3ee6afbe0211d,
+ 0xb8157268fdae9e4c, 0x5960ea05bad82964,
+ 0xe61acf033d1a45df, 0x6fb92487298e33bd,
+ 0x8fd0c16206306bab, 0xa5d3b6d479f8e056,
+ 0xb3c4f1ba87bc8696, 0x8f48a4899877186c,
+ 0xe0b62e2929aba83c, 0x331acdabfe94de87,
+ 0x8c71dcd9ba0b4925, 0x9ff0c08b7f1d0b14,
+ 0xaf8e5410288e1b6f, 0x7ecf0ae5ee44dd9,
+ 0xdb71e91432b1a24a, 0xc9e82cd9f69d6150,
+ 0x892731ac9faf056e, 0xbe311c083a225cd2,
+ 0xab70fe17c79ac6ca, 0x6dbd630a48aaf406,
+ 0xd64d3d9db981787d, 0x92cbbccdad5b108,
+ 0x85f0468293f0eb4e, 0x25bbf56008c58ea5,
+ 0xa76c582338ed2621, 0xaf2af2b80af6f24e,
+ 0xd1476e2c07286faa, 0x1af5af660db4aee1,
+ 0x82cca4db847945ca, 0x50d98d9fc890ed4d,
+ 0xa37fce126597973c, 0xe50ff107bab528a0,
+ 0xcc5fc196fefd7d0c, 0x1e53ed49a96272c8,
+ 0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7a,
+ 0x9faacf3df73609b1, 0x77b191618c54e9ac,
+ 0xc795830d75038c1d, 0xd59df5b9ef6a2417,
+ 0xf97ae3d0d2446f25, 0x4b0573286b44ad1d,
+ 0x9becce62836ac577, 0x4ee367f9430aec32,
+ 0xc2e801fb244576d5, 0x229c41f793cda73f,
+ 0xf3a20279ed56d48a, 0x6b43527578c1110f,
+ 0x9845418c345644d6, 0x830a13896b78aaa9,
+ 0xbe5691ef416bd60c, 0x23cc986bc656d553,
+ 0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa8,
+ 0x94b3a202eb1c3f39, 0x7bf7d71432f3d6a9,
+ 0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc53,
+ 0xe858ad248f5c22c9, 0xd1b3400f8f9cff68,
+ 0x91376c36d99995be, 0x23100809b9c21fa1,
+ 0xb58547448ffffb2d, 0xabd40a0c2832a78a,
+ 0xe2e69915b3fff9f9, 0x16c90c8f323f516c,
+ 0x8dd01fad907ffc3b, 0xae3da7d97f6792e3,
+ 0xb1442798f49ffb4a, 0x99cd11cfdf41779c,
+ 0xdd95317f31c7fa1d, 0x40405643d711d583,
+ 0x8a7d3eef7f1cfc52, 0x482835ea666b2572,
+ 0xad1c8eab5ee43b66, 0xda3243650005eecf,
+ 0xd863b256369d4a40, 0x90bed43e40076a82,
+ 0x873e4f75e2224e68, 0x5a7744a6e804a291,
+ 0xa90de3535aaae202, 0x711515d0a205cb36,
+ 0xd3515c2831559a83, 0xd5a5b44ca873e03,
+ 0x8412d9991ed58091, 0xe858790afe9486c2,
+ 0xa5178fff668ae0b6, 0x626e974dbe39a872,
+ 0xce5d73ff402d98e3, 0xfb0a3d212dc8128f,
+ 0x80fa687f881c7f8e, 0x7ce66634bc9d0b99,
+ 0xa139029f6a239f72, 0x1c1fffc1ebc44e80,
+ 0xc987434744ac874e, 0xa327ffb266b56220,
+ 0xfbe9141915d7a922, 0x4bf1ff9f0062baa8,
+ 0x9d71ac8fada6c9b5, 0x6f773fc3603db4a9,
+ 0xc4ce17b399107c22, 0xcb550fb4384d21d3,
+ 0xf6019da07f549b2b, 0x7e2a53a146606a48,
+ 0x99c102844f94e0fb, 0x2eda7444cbfc426d,
+ 0xc0314325637a1939, 0xfa911155fefb5308,
+ 0xf03d93eebc589f88, 0x793555ab7eba27ca,
+ 0x96267c7535b763b5, 0x4bc1558b2f3458de,
+ 0xbbb01b9283253ca2, 0x9eb1aaedfb016f16,
+ 0xea9c227723ee8bcb, 0x465e15a979c1cadc,
+ 0x92a1958a7675175f, 0xbfacd89ec191ec9,
+ 0xb749faed14125d36, 0xcef980ec671f667b,
+ 0xe51c79a85916f484, 0x82b7e12780e7401a,
+ 0x8f31cc0937ae58d2, 0xd1b2ecb8b0908810,
+ 0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa15,
+ 0xdfbdcece67006ac9, 0x67a791e093e1d49a,
+ 0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e0,
+ 0xaecc49914078536d, 0x58fae9f773886e18,
+ 0xda7f5bf590966848, 0xaf39a475506a899e,
+ 0x888f99797a5e012d, 0x6d8406c952429603,
+ 0xaab37fd7d8f58178, 0xc8e5087ba6d33b83,
+ 0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a64,
+ 0x855c3be0a17fcd26, 0x5cf2eea09a55067f,
+ 0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481e,
+ 0xd0601d8efc57b08b, 0xf13b94daf124da26,
+ 0x823c12795db6ce57, 0x76c53d08d6b70858,
+ 0xa2cb1717b52481ed, 0x54768c4b0c64ca6e,
+ 0xcb7ddcdda26da268, 0xa9942f5dcf7dfd09,
+ 0xfe5d54150b090b02, 0xd3f93b35435d7c4c,
+ 0x9efa548d26e5a6e1, 0xc47bc5014a1a6daf,
+ 0xc6b8e9b0709f109a, 0x359ab6419ca1091b,
+ 0xf867241c8cc6d4c0, 0xc30163d203c94b62,
+ 0x9b407691d7fc44f8, 0x79e0de63425dcf1d,
+ 0xc21094364dfb5636, 0x985915fc12f542e4,
+ 0xf294b943e17a2bc4, 0x3e6f5b7b17b2939d,
+ 0x979cf3ca6cec5b5a, 0xa705992ceecf9c42,
+ 0xbd8430bd08277231, 0x50c6ff782a838353,
+ 0xece53cec4a314ebd, 0xa4f8bf5635246428,
+ 0x940f4613ae5ed136, 0x871b7795e136be99,
+ 0xb913179899f68584, 0x28e2557b59846e3f,
+ 0xe757dd7ec07426e5, 0x331aeada2fe589cf,
+ 0x9096ea6f3848984f, 0x3ff0d2c85def7621,
+ 0xb4bca50b065abe63, 0xfed077a756b53a9,
+ 0xe1ebce4dc7f16dfb, 0xd3e8495912c62894,
+ 0x8d3360f09cf6e4bd, 0x64712dd7abbbd95c,
+ 0xb080392cc4349dec, 0xbd8d794d96aacfb3,
+ 0xdca04777f541c567, 0xecf0d7a0fc5583a0,
+ 0x89e42caaf9491b60, 0xf41686c49db57244,
+ 0xac5d37d5b79b6239, 0x311c2875c522ced5,
+ 0xd77485cb25823ac7, 0x7d633293366b828b,
+ 0x86a8d39ef77164bc, 0xae5dff9c02033197,
+ 0xa8530886b54dbdeb, 0xd9f57f830283fdfc,
+ 0xd267caa862a12d66, 0xd072df63c324fd7b,
+ 0x8380dea93da4bc60, 0x4247cb9e59f71e6d,
+ 0xa46116538d0deb78, 0x52d9be85f074e608,
+ 0xcd795be870516656, 0x67902e276c921f8b,
+ 0x806bd9714632dff6, 0xba1cd8a3db53b6,
+ 0xa086cfcd97bf97f3, 0x80e8a40eccd228a4,
+ 0xc8a883c0fdaf7df0, 0x6122cd128006b2cd,
+ 0xfad2a4b13d1b5d6c, 0x796b805720085f81,
+ 0x9cc3a6eec6311a63, 0xcbe3303674053bb0,
+ 0xc3f490aa77bd60fc, 0xbedbfc4411068a9c,
+ 0xf4f1b4d515acb93b, 0xee92fb5515482d44,
+ 0x991711052d8bf3c5, 0x751bdd152d4d1c4a,
+ 0xbf5cd54678eef0b6, 0xd262d45a78a0635d,
+ 0xef340a98172aace4, 0x86fb897116c87c34,
+ 0x9580869f0e7aac0e, 0xd45d35e6ae3d4da0,
+ 0xbae0a846d2195712, 0x8974836059cca109,
+ 0xe998d258869facd7, 0x2bd1a438703fc94b,
+ 0x91ff83775423cc06, 0x7b6306a34627ddcf,
+ 0xb67f6455292cbf08, 0x1a3bc84c17b1d542,
+ 0xe41f3d6a7377eeca, 0x20caba5f1d9e4a93,
+ 0x8e938662882af53e, 0x547eb47b7282ee9c,
+ 0xb23867fb2a35b28d, 0xe99e619a4f23aa43,
+ 0xdec681f9f4c31f31, 0x6405fa00e2ec94d4,
+ 0x8b3c113c38f9f37e, 0xde83bc408dd3dd04,
+ 0xae0b158b4738705e, 0x9624ab50b148d445,
+ 0xd98ddaee19068c76, 0x3badd624dd9b0957,
+ 0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d6,
+ 0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4c,
+ 0xd47487cc8470652b, 0x7647c3200069671f,
+ 0x84c8d4dfd2c63f3b, 0x29ecd9f40041e073,
+ 0xa5fb0a17c777cf09, 0xf468107100525890,
+ 0xcf79cc9db955c2cc, 0x7182148d4066eeb4,
+ 0x81ac1fe293d599bf, 0xc6f14cd848405530,
+ 0xa21727db38cb002f, 0xb8ada00e5a506a7c,
+ 0xca9cf1d206fdc03b, 0xa6d90811f0e4851c,
+ 0xfd442e4688bd304a, 0x908f4a166d1da663,
+ 0x9e4a9cec15763e2e, 0x9a598e4e043287fe,
+ 0xc5dd44271ad3cdba, 0x40eff1e1853f29fd,
+ 0xf7549530e188c128, 0xd12bee59e68ef47c,
+ 0x9a94dd3e8cf578b9, 0x82bb74f8301958ce,
+ 0xc13a148e3032d6e7, 0xe36a52363c1faf01,
+ 0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac1,
+ 0x96f5600f15a7b7e5, 0x29ab103a5ef8c0b9,
+ 0xbcb2b812db11a5de, 0x7415d448f6b6f0e7,
+ 0xebdf661791d60f56, 0x111b495b3464ad21,
+ 0x936b9fcebb25c995, 0xcab10dd900beec34,
+ 0xb84687c269ef3bfb, 0x3d5d514f40eea742,
+ 0xe65829b3046b0afa, 0xcb4a5a3112a5112,
+ 0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ab,
+ 0xb3f4e093db73a093, 0x59ed216765690f56,
+ 0xe0f218b8d25088b8, 0x306869c13ec3532c,
+ 0x8c974f7383725573, 0x1e414218c73a13fb,
+ 0xafbd2350644eeacf, 0xe5d1929ef90898fa,
+ 0xdbac6c247d62a583, 0xdf45f746b74abf39,
+ 0x894bc396ce5da772, 0x6b8bba8c328eb783,
+ 0xab9eb47c81f5114f, 0x66ea92f3f326564,
+ 0xd686619ba27255a2, 0xc80a537b0efefebd,
+ 0x8613fd0145877585, 0xbd06742ce95f5f36,
+ 0xa798fc4196e952e7, 0x2c48113823b73704,
+ 0xd17f3b51fca3a7a0, 0xf75a15862ca504c5,
+ 0x82ef85133de648c4, 0x9a984d73dbe722fb,
+ 0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebba,
+ 0xcc963fee10b7d1b3, 0x318df905079926a8,
+ 0xffbbcfe994e5c61f, 0xfdf17746497f7052,
+ 0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa633,
+ 0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc0,
+ 0xf9bd690a1b68637b, 0x3dfdce7aa3c673b0,
+ 0x9c1661a651213e2d, 0x6bea10ca65c084e,
+ 0xc31bfa0fe5698db8, 0x486e494fcff30a62,
+ 0xf3e2f893dec3f126, 0x5a89dba3c3efccfa,
+ 0x986ddb5c6b3a76b7, 0xf89629465a75e01c,
+ 0xbe89523386091465, 0xf6bbb397f1135823,
+ 0xee2ba6c0678b597f, 0x746aa07ded582e2c,
+ 0x94db483840b717ef, 0xa8c2a44eb4571cdc,
+ 0xba121a4650e4ddeb, 0x92f34d62616ce413,
+ 0xe896a0d7e51e1566, 0x77b020baf9c81d17,
+ 0x915e2486ef32cd60, 0xace1474dc1d122e,
+ 0xb5b5ada8aaff80b8, 0xd819992132456ba,
+ 0xe3231912d5bf60e6, 0x10e1fff697ed6c69,
+ 0x8df5efabc5979c8f, 0xca8d3ffa1ef463c1,
+ 0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb2,
+ 0xddd0467c64bce4a0, 0xac7cb3f6d05ddbde,
+ 0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96b,
+ 0xad4ab7112eb3929d, 0x86c16c98d2c953c6,
+ 0xd89d64d57a607744, 0xe871c7bf077ba8b7,
+ 0x87625f056c7c4a8b, 0x11471cd764ad4972,
+ 0xa93af6c6c79b5d2d, 0xd598e40d3dd89bcf,
+ 0xd389b47879823479, 0x4aff1d108d4ec2c3,
+ 0x843610cb4bf160cb, 0xcedf722a585139ba,
+ 0xa54394fe1eedb8fe, 0xc2974eb4ee658828,
+ 0xce947a3da6a9273e, 0x733d226229feea32,
+ 0x811ccc668829b887, 0x806357d5a3f525f,
+ 0xa163ff802a3426a8, 0xca07c2dcb0cf26f7,
+ 0xc9bcff6034c13052, 0xfc89b393dd02f0b5,
+ 0xfc2c3f3841f17c67, 0xbbac2078d443ace2,
+ 0x9d9ba7832936edc0, 0xd54b944b84aa4c0d,
+ 0xc5029163f384a931, 0xa9e795e65d4df11,
+ 0xf64335bcf065d37d, 0x4d4617b5ff4a16d5,
+ 0x99ea0196163fa42e, 0x504bced1bf8e4e45,
+ 0xc06481fb9bcf8d39, 0xe45ec2862f71e1d6,
+ 0xf07da27a82c37088, 0x5d767327bb4e5a4c,
+ 0x964e858c91ba2655, 0x3a6a07f8d510f86f,
+ 0xbbe226efb628afea, 0x890489f70a55368b,
+ 0xeadab0aba3b2dbe5, 0x2b45ac74ccea842e,
+ 0x92c8ae6b464fc96f, 0x3b0b8bc90012929d,
+ 0xb77ada0617e3bbcb, 0x9ce6ebb40173744,
+ 0xe55990879ddcaabd, 0xcc420a6a101d0515,
+ 0x8f57fa54c2a9eab6, 0x9fa946824a12232d,
+ 0xb32df8e9f3546564, 0x47939822dc96abf9,
+ 0xdff9772470297ebd, 0x59787e2b93bc56f7,
+ 0x8bfbea76c619ef36, 0x57eb4edb3c55b65a,
+ 0xaefae51477a06b03, 0xede622920b6b23f1,
+ 0xdab99e59958885c4, 0xe95fab368e45eced,
+ 0x88b402f7fd75539b, 0x11dbcb0218ebb414,
+ 0xaae103b5fcd2a881, 0xd652bdc29f26a119,
+ 0xd59944a37c0752a2, 0x4be76d3346f0495f,
+ 0x857fcae62d8493a5, 0x6f70a4400c562ddb,
+ 0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb952,
+ 0xd097ad07a71f26b2, 0x7e2000a41346a7a7,
+ 0x825ecc24c873782f, 0x8ed400668c0c28c8,
+ 0xa2f67f2dfa90563b, 0x728900802f0f32fa,
+ 0xcbb41ef979346bca, 0x4f2b40a03ad2ffb9,
+ 0xfea126b7d78186bc, 0xe2f610c84987bfa8,
+ 0x9f24b832e6b0f436, 0xdd9ca7d2df4d7c9,
+ 0xc6ede63fa05d3143, 0x91503d1c79720dbb,
+ 0xf8a95fcf88747d94, 0x75a44c6397ce912a,
+ 0x9b69dbe1b548ce7c, 0xc986afbe3ee11aba,
+ 0xc24452da229b021b, 0xfbe85badce996168,
+ 0xf2d56790ab41c2a2, 0xfae27299423fb9c3,
+ 0x97c560ba6b0919a5, 0xdccd879fc967d41a,
+ 0xbdb6b8e905cb600f, 0x5400e987bbc1c920,
+ 0xed246723473e3813, 0x290123e9aab23b68,
+ 0x9436c0760c86e30b, 0xf9a0b6720aaf6521,
+ 0xb94470938fa89bce, 0xf808e40e8d5b3e69,
+ 0xe7958cb87392c2c2, 0xb60b1d1230b20e04,
+ 0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c2,
+ 0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af3,
+ 0xe2280b6c20dd5232, 0x25c6da63c38de1b0,
+ 0x8d590723948a535f, 0x579c487e5a38ad0e,
+ 0xb0af48ec79ace837, 0x2d835a9df0c6d851,
+ 0xdcdb1b2798182244, 0xf8e431456cf88e65,
+ 0x8a08f0f8bf0f156b, 0x1b8e9ecb641b58ff,
+ 0xac8b2d36eed2dac5, 0xe272467e3d222f3f,
+ 0xd7adf884aa879177, 0x5b0ed81dcc6abb0f,
+ 0x86ccbb52ea94baea, 0x98e947129fc2b4e9,
+ 0xa87fea27a539e9a5, 0x3f2398d747b36224,
+ 0xd29fe4b18e88640e, 0x8eec7f0d19a03aad,
+ 0x83a3eeeef9153e89, 0x1953cf68300424ac,
+ 0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd7,
+ 0xcdb02555653131b6, 0x3792f412cb06794d,
+ 0x808e17555f3ebf11, 0xe2bbd88bbee40bd0,
+ 0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec4,
+ 0xc8de047564d20a8b, 0xf245825a5a445275,
+ 0xfb158592be068d2e, 0xeed6e2f0f0d56712,
+ 0x9ced737bb6c4183d, 0x55464dd69685606b,
+ 0xc428d05aa4751e4c, 0xaa97e14c3c26b886,
+ 0xf53304714d9265df, 0xd53dd99f4b3066a8,
+ 0x993fe2c6d07b7fab, 0xe546a8038efe4029,
+ 0xbf8fdb78849a5f96, 0xde98520472bdd033,
+ 0xef73d256a5c0f77c, 0x963e66858f6d4440,
+ 0x95a8637627989aad, 0xdde7001379a44aa8,
+ 0xbb127c53b17ec159, 0x5560c018580d5d52,
+ 0xe9d71b689dde71af, 0xaab8f01e6e10b4a6,
+ 0x9226712162ab070d, 0xcab3961304ca70e8,
+ 0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d22,
+ 0xe45c10c42a2b3b05, 0x8cb89a7db77c506a,
+ 0x8eb98a7a9a5b04e3, 0x77f3608e92adb242,
+ 0xb267ed1940f1c61c, 0x55f038b237591ed3,
+ 0xdf01e85f912e37a3, 0x6b6c46dec52f6688,
+ 0x8b61313bbabce2c6, 0x2323ac4b3b3da015,
+ 0xae397d8aa96c1b77, 0xabec975e0a0d081a,
+ 0xd9c7dced53c72255, 0x96e7bd358c904a21,
+ 0x881cea14545c7575, 0x7e50d64177da2e54,
+ 0xaa242499697392d2, 0xdde50bd1d5d0b9e9,
+ 0xd4ad2dbfc3d07787, 0x955e4ec64b44e864,
+ 0x84ec3c97da624ab4, 0xbd5af13bef0b113e,
+ 0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58e,
+ 0xcfb11ead453994ba, 0x67de18eda5814af2,
+ 0x81ceb32c4b43fcf4, 0x80eacf948770ced7,
+ 0xa2425ff75e14fc31, 0xa1258379a94d028d,
+ 0xcad2f7f5359a3b3e, 0x96ee45813a04330,
+ 0xfd87b5f28300ca0d, 0x8bca9d6e188853fc,
+ 0x9e74d1b791e07e48, 0x775ea264cf55347e,
+ 0xc612062576589dda, 0x95364afe032a819e,
+ 0xf79687aed3eec551, 0x3a83ddbd83f52205,
+ 0x9abe14cd44753b52, 0xc4926a9672793543,
+ 0xc16d9a0095928a27, 0x75b7053c0f178294,
+ 0xf1c90080baf72cb1, 0x5324c68b12dd6339,
+ 0x971da05074da7bee, 0xd3f6fc16ebca5e04,
+ 0xbce5086492111aea, 0x88f4bb1ca6bcf585,
+ 0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6,
+ 0x9392ee8e921d5d07, 0x3aff322e62439fd0,
+ 0xb877aa3236a4b449, 0x9befeb9fad487c3,
+ 0xe69594bec44de15b, 0x4c2ebe687989a9b4,
+ 0x901d7cf73ab0acd9, 0xf9d37014bf60a11,
+ 0xb424dc35095cd80f, 0x538484c19ef38c95,
+ 0xe12e13424bb40e13, 0x2865a5f206b06fba,
+ 0x8cbccc096f5088cb, 0xf93f87b7442e45d4,
+ 0xafebff0bcb24aafe, 0xf78f69a51539d749,
+ 0xdbe6fecebdedd5be, 0xb573440e5a884d1c,
+ 0x89705f4136b4a597, 0x31680a88f8953031,
+ 0xabcc77118461cefc, 0xfdc20d2b36ba7c3e,
+ 0xd6bf94d5e57a42bc, 0x3d32907604691b4d,
+ 0x8637bd05af6c69b5, 0xa63f9a49c2c1b110,
+ 0xa7c5ac471b478423, 0xfcf80dc33721d54,
+ 0xd1b71758e219652b, 0xd3c36113404ea4a9,
+ 0x83126e978d4fdf3b, 0x645a1cac083126ea,
+ 0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4,
+ 0xcccccccccccccccc, 0xcccccccccccccccd,
+ 0x8000000000000000, 0x0,
+ 0xa000000000000000, 0x0,
+ 0xc800000000000000, 0x0,
+ 0xfa00000000000000, 0x0,
+ 0x9c40000000000000, 0x0,
+ 0xc350000000000000, 0x0,
+ 0xf424000000000000, 0x0,
+ 0x9896800000000000, 0x0,
+ 0xbebc200000000000, 0x0,
+ 0xee6b280000000000, 0x0,
+ 0x9502f90000000000, 0x0,
+ 0xba43b74000000000, 0x0,
+ 0xe8d4a51000000000, 0x0,
+ 0x9184e72a00000000, 0x0,
+ 0xb5e620f480000000, 0x0,
+ 0xe35fa931a0000000, 0x0,
+ 0x8e1bc9bf04000000, 0x0,
+ 0xb1a2bc2ec5000000, 0x0,
+ 0xde0b6b3a76400000, 0x0,
+ 0x8ac7230489e80000, 0x0,
+ 0xad78ebc5ac620000, 0x0,
+ 0xd8d726b7177a8000, 0x0,
+ 0x878678326eac9000, 0x0,
+ 0xa968163f0a57b400, 0x0,
+ 0xd3c21bcecceda100, 0x0,
+ 0x84595161401484a0, 0x0,
+ 0xa56fa5b99019a5c8, 0x0,
+ 0xcecb8f27f4200f3a, 0x0,
+ 0x813f3978f8940984, 0x4000000000000000,
+ 0xa18f07d736b90be5, 0x5000000000000000,
+ 0xc9f2c9cd04674ede, 0xa400000000000000,
+ 0xfc6f7c4045812296, 0x4d00000000000000,
+ 0x9dc5ada82b70b59d, 0xf020000000000000,
+ 0xc5371912364ce305, 0x6c28000000000000,
+ 0xf684df56c3e01bc6, 0xc732000000000000,
+ 0x9a130b963a6c115c, 0x3c7f400000000000,
+ 0xc097ce7bc90715b3, 0x4b9f100000000000,
+ 0xf0bdc21abb48db20, 0x1e86d40000000000,
+ 0x96769950b50d88f4, 0x1314448000000000,
+ 0xbc143fa4e250eb31, 0x17d955a000000000,
+ 0xeb194f8e1ae525fd, 0x5dcfab0800000000,
+ 0x92efd1b8d0cf37be, 0x5aa1cae500000000,
+ 0xb7abc627050305ad, 0xf14a3d9e40000000,
+ 0xe596b7b0c643c719, 0x6d9ccd05d0000000,
+ 0x8f7e32ce7bea5c6f, 0xe4820023a2000000,
+ 0xb35dbf821ae4f38b, 0xdda2802c8a800000,
+ 0xe0352f62a19e306e, 0xd50b2037ad200000,
+ 0x8c213d9da502de45, 0x4526f422cc340000,
+ 0xaf298d050e4395d6, 0x9670b12b7f410000,
+ 0xdaf3f04651d47b4c, 0x3c0cdd765f114000,
+ 0x88d8762bf324cd0f, 0xa5880a69fb6ac800,
+ 0xab0e93b6efee0053, 0x8eea0d047a457a00,
+ 0xd5d238a4abe98068, 0x72a4904598d6d880,
+ 0x85a36366eb71f041, 0x47a6da2b7f864750,
+ 0xa70c3c40a64e6c51, 0x999090b65f67d924,
+ 0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d,
+ 0x82818f1281ed449f, 0xbff8f10e7a8921a4,
+ 0xa321f2d7226895c7, 0xaff72d52192b6a0d,
+ 0xcbea6f8ceb02bb39, 0x9bf4f8a69f764490,
+ 0xfee50b7025c36a08, 0x2f236d04753d5b4,
+ 0x9f4f2726179a2245, 0x1d762422c946590,
+ 0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef5,
+ 0xf8ebad2b84e0d58b, 0xd2e0898765a7deb2,
+ 0x9b934c3b330c8577, 0x63cc55f49f88eb2f,
+ 0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fb,
+ 0xf316271c7fc3908a, 0x8bef464e3945ef7a,
+ 0x97edd871cfda3a56, 0x97758bf0e3cbb5ac,
+ 0xbde94e8e43d0c8ec, 0x3d52eeed1cbea317,
+ 0xed63a231d4c4fb27, 0x4ca7aaa863ee4bdd,
+ 0x945e455f24fb1cf8, 0x8fe8caa93e74ef6a,
+ 0xb975d6b6ee39e436, 0xb3e2fd538e122b44,
+ 0xe7d34c64a9c85d44, 0x60dbbca87196b616,
+ 0x90e40fbeea1d3a4a, 0xbc8955e946fe31cd,
+ 0xb51d13aea4a488dd, 0x6babab6398bdbe41,
+ 0xe264589a4dcdab14, 0xc696963c7eed2dd1,
+ 0x8d7eb76070a08aec, 0xfc1e1de5cf543ca2,
+ 0xb0de65388cc8ada8, 0x3b25a55f43294bcb,
+ 0xdd15fe86affad912, 0x49ef0eb713f39ebe,
+ 0x8a2dbf142dfcc7ab, 0x6e3569326c784337,
+ 0xacb92ed9397bf996, 0x49c2c37f07965404,
+ 0xd7e77a8f87daf7fb, 0xdc33745ec97be906,
+ 0x86f0ac99b4e8dafd, 0x69a028bb3ded71a3,
+ 0xa8acd7c0222311bc, 0xc40832ea0d68ce0c,
+ 0xd2d80db02aabd62b, 0xf50a3fa490c30190,
+ 0x83c7088e1aab65db, 0x792667c6da79e0fa,
+ 0xa4b8cab1a1563f52, 0x577001b891185938,
+ 0xcde6fd5e09abcf26, 0xed4c0226b55e6f86,
+ 0x80b05e5ac60b6178, 0x544f8158315b05b4,
+ 0xa0dc75f1778e39d6, 0x696361ae3db1c721,
+ 0xc913936dd571c84c, 0x3bc3a19cd1e38e9,
+ 0xfb5878494ace3a5f, 0x4ab48a04065c723,
+ 0x9d174b2dcec0e47b, 0x62eb0d64283f9c76,
+ 0xc45d1df942711d9a, 0x3ba5d0bd324f8394,
+ 0xf5746577930d6500, 0xca8f44ec7ee36479,
+ 0x9968bf6abbe85f20, 0x7e998b13cf4e1ecb,
+ 0xbfc2ef456ae276e8, 0x9e3fedd8c321a67e,
+ 0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101e,
+ 0x95d04aee3b80ece5, 0xbba1f1d158724a12,
+ 0xbb445da9ca61281f, 0x2a8a6e45ae8edc97,
+ 0xea1575143cf97226, 0xf52d09d71a3293bd,
+ 0x924d692ca61be758, 0x593c2626705f9c56,
+ 0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836c,
+ 0xe498f455c38b997a, 0xb6dfb9c0f956447,
+ 0x8edf98b59a373fec, 0x4724bd4189bd5eac,
+ 0xb2977ee300c50fe7, 0x58edec91ec2cb657,
+ 0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ed,
+ 0x8b865b215899f46c, 0xbd79e0d20082ee74,
+ 0xae67f1e9aec07187, 0xecd8590680a3aa11,
+ 0xda01ee641a708de9, 0xe80e6f4820cc9495,
+ 0x884134fe908658b2, 0x3109058d147fdcdd,
+ 0xaa51823e34a7eede, 0xbd4b46f0599fd415,
+ 0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91a,
+ 0x850fadc09923329e, 0x3e2cf6bc604ddb0,
+ 0xa6539930bf6bff45, 0x84db8346b786151c,
+ 0xcfe87f7cef46ff16, 0xe612641865679a63,
+ 0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07e,
+ 0xa26da3999aef7749, 0xe3be5e330f38f09d,
+ 0xcb090c8001ab551c, 0x5cadf5bfd3072cc5,
+ 0xfdcb4fa002162a63, 0x73d9732fc7c8f7f6,
+ 0x9e9f11c4014dda7e, 0x2867e7fddcdd9afa,
+ 0xc646d63501a1511d, 0xb281e1fd541501b8,
+ 0xf7d88bc24209a565, 0x1f225a7ca91a4226,
+ 0x9ae757596946075f, 0x3375788de9b06958,
+ 0xc1a12d2fc3978937, 0x52d6b1641c83ae,
+ 0xf209787bb47d6b84, 0xc0678c5dbd23a49a,
+ 0x9745eb4d50ce6332, 0xf840b7ba963646e0,
+ 0xbd176620a501fbff, 0xb650e5a93bc3d898,
+ 0xec5d3fa8ce427aff, 0xa3e51f138ab4cebe,
+ 0x93ba47c980e98cdf, 0xc66f336c36b10137,
+ 0xb8a8d9bbe123f017, 0xb80b0047445d4184,
+ 0xe6d3102ad96cec1d, 0xa60dc059157491e5,
+ 0x9043ea1ac7e41392, 0x87c89837ad68db2f,
+ 0xb454e4a179dd1877, 0x29babe4598c311fb,
+ 0xe16a1dc9d8545e94, 0xf4296dd6fef3d67a,
+ 0x8ce2529e2734bb1d, 0x1899e4a65f58660c,
+ 0xb01ae745b101e9e4, 0x5ec05dcff72e7f8f,
+ 0xdc21a1171d42645d, 0x76707543f4fa1f73,
+ 0x899504ae72497eba, 0x6a06494a791c53a8,
+ 0xabfa45da0edbde69, 0x487db9d17636892,
+ 0xd6f8d7509292d603, 0x45a9d2845d3c42b6,
+ 0x865b86925b9bc5c2, 0xb8a2392ba45a9b2,
+ 0xa7f26836f282b732, 0x8e6cac7768d7141e,
+ 0xd1ef0244af2364ff, 0x3207d795430cd926,
+ 0x8335616aed761f1f, 0x7f44e6bd49e807b8,
+ 0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a6,
+ 0xcd036837130890a1, 0x36dba887c37a8c0f,
+ 0x802221226be55a64, 0xc2494954da2c9789,
+ 0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6c,
+ 0xc83553c5c8965d3d, 0x6f92829494e5acc7,
+ 0xfa42a8b73abbf48c, 0xcb772339ba1f17f9,
+ 0x9c69a97284b578d7, 0xff2a760414536efb,
+ 0xc38413cf25e2d70d, 0xfef5138519684aba,
+ 0xf46518c2ef5b8cd1, 0x7eb258665fc25d69,
+ 0x98bf2f79d5993802, 0xef2f773ffbd97a61,
+ 0xbeeefb584aff8603, 0xaafb550ffacfd8fa,
+ 0xeeaaba2e5dbf6784, 0x95ba2a53f983cf38,
+ 0x952ab45cfa97a0b2, 0xdd945a747bf26183,
+ 0xba756174393d88df, 0x94f971119aeef9e4,
+ 0xe912b9d1478ceb17, 0x7a37cd5601aab85d,
+ 0x91abb422ccb812ee, 0xac62e055c10ab33a,
+ 0xb616a12b7fe617aa, 0x577b986b314d6009,
+ 0xe39c49765fdf9d94, 0xed5a7e85fda0b80b,
+ 0x8e41ade9fbebc27d, 0x14588f13be847307,
+ 0xb1d219647ae6b31c, 0x596eb2d8ae258fc8,
+ 0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bb,
+ 0x8aec23d680043bee, 0x25de7bb9480d5854,
+ 0xada72ccc20054ae9, 0xaf561aa79a10ae6a,
+ 0xd910f7ff28069da4, 0x1b2ba1518094da04,
+ 0x87aa9aff79042286, 0x90fb44d2f05d0842,
+ 0xa99541bf57452b28, 0x353a1607ac744a53,
+ 0xd3fa922f2d1675f2, 0x42889b8997915ce8,
+ 0x847c9b5d7c2e09b7, 0x69956135febada11,
+ 0xa59bc234db398c25, 0x43fab9837e699095,
+ 0xcf02b2c21207ef2e, 0x94f967e45e03f4bb,
+ 0x8161afb94b44f57d, 0x1d1be0eebac278f5,
+ 0xa1ba1ba79e1632dc, 0x6462d92a69731732,
+ 0xca28a291859bbf93, 0x7d7b8f7503cfdcfe,
+ 0xfcb2cb35e702af78, 0x5cda735244c3d43e,
+ 0x9defbf01b061adab, 0x3a0888136afa64a7,
+ 0xc56baec21c7a1916, 0x88aaa1845b8fdd0,
+ 0xf6c69a72a3989f5b, 0x8aad549e57273d45,
+ 0x9a3c2087a63f6399, 0x36ac54e2f678864b,
+ 0xc0cb28a98fcf3c7f, 0x84576a1bb416a7dd,
+ 0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d5,
+ 0x969eb7c47859e743, 0x9f644ae5a4b1b325,
+ 0xbc4665b596706114, 0x873d5d9f0dde1fee,
+ 0xeb57ff22fc0c7959, 0xa90cb506d155a7ea,
+ 0x9316ff75dd87cbd8, 0x9a7f12442d588f2,
+ 0xb7dcbf5354e9bece, 0xc11ed6d538aeb2f,
+ 0xe5d3ef282a242e81, 0x8f1668c8a86da5fa,
+ 0x8fa475791a569d10, 0xf96e017d694487bc,
+ 0xb38d92d760ec4455, 0x37c981dcc395a9ac,
+ 0xe070f78d3927556a, 0x85bbe253f47b1417,
+ 0x8c469ab843b89562, 0x93956d7478ccec8e,
+ 0xaf58416654a6babb, 0x387ac8d1970027b2,
+ 0xdb2e51bfe9d0696a, 0x6997b05fcc0319e,
+ 0x88fcf317f22241e2, 0x441fece3bdf81f03,
+ 0xab3c2fddeeaad25a, 0xd527e81cad7626c3,
+ 0xd60b3bd56a5586f1, 0x8a71e223d8d3b074,
+ 0x85c7056562757456, 0xf6872d5667844e49,
+ 0xa738c6bebb12d16c, 0xb428f8ac016561db,
+ 0xd106f86e69d785c7, 0xe13336d701beba52,
+ 0x82a45b450226b39c, 0xecc0024661173473,
+ 0xa34d721642b06084, 0x27f002d7f95d0190,
+ 0xcc20ce9bd35c78a5, 0x31ec038df7b441f4,
+ 0xff290242c83396ce, 0x7e67047175a15271,
+ 0x9f79a169bd203e41, 0xf0062c6e984d386,
+ 0xc75809c42c684dd1, 0x52c07b78a3e60868,
+ 0xf92e0c3537826145, 0xa7709a56ccdf8a82,
+ 0x9bbcc7a142b17ccb, 0x88a66076400bb691,
+ 0xc2abf989935ddbfe, 0x6acff893d00ea435,
+ 0xf356f7ebf83552fe, 0x583f6b8c4124d43,
+ 0x98165af37b2153de, 0xc3727a337a8b704a,
+ 0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5c,
+ 0xeda2ee1c7064130c, 0x1162def06f79df73,
+ 0x9485d4d1c63e8be7, 0x8addcb5645ac2ba8,
+ 0xb9a74a0637ce2ee1, 0x6d953e2bd7173692,
+ 0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0437,
+ 0x910ab1d4db9914a0, 0x1d9c9892400a22a2,
+ 0xb54d5e4a127f59c8, 0x2503beb6d00cab4b,
+ 0xe2a0b5dc971f303a, 0x2e44ae64840fd61d,
+ 0x8da471a9de737e24, 0x5ceaecfed289e5d2,
+ 0xb10d8e1456105dad, 0x7425a83e872c5f47,
+ 0xdd50f1996b947518, 0xd12f124e28f77719,
+ 0x8a5296ffe33cc92f, 0x82bd6b70d99aaa6f,
+ 0xace73cbfdc0bfb7b, 0x636cc64d1001550b,
+ 0xd8210befd30efa5a, 0x3c47f7e05401aa4e,
+ 0x8714a775e3e95c78, 0x65acfaec34810a71,
+ 0xa8d9d1535ce3b396, 0x7f1839a741a14d0d,
+ 0xd31045a8341ca07c, 0x1ede48111209a050,
+ 0x83ea2b892091e44d, 0x934aed0aab460432,
+ 0xa4e4b66b68b65d60, 0xf81da84d5617853f,
+ 0xce1de40642e3f4b9, 0x36251260ab9d668e,
+ 0x80d2ae83e9ce78f3, 0xc1d72b7c6b426019,
+ 0xa1075a24e4421730, 0xb24cf65b8612f81f,
+ 0xc94930ae1d529cfc, 0xdee033f26797b627,
+ 0xfb9b7cd9a4a7443c, 0x169840ef017da3b1,
+ 0x9d412e0806e88aa5, 0x8e1f289560ee864e,
+ 0xc491798a08a2ad4e, 0xf1a6f2bab92a27e2,
+ 0xf5b5d7ec8acb58a2, 0xae10af696774b1db,
+ 0x9991a6f3d6bf1765, 0xacca6da1e0a8ef29,
+ 0xbff610b0cc6edd3f, 0x17fd090a58d32af3,
+ 0xeff394dcff8a948e, 0xddfc4b4cef07f5b0,
+ 0x95f83d0a1fb69cd9, 0x4abdaf101564f98e,
+ 0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f1,
+ 0xea53df5fd18d5513, 0x84c86189216dc5ed,
+ 0x92746b9be2f8552c, 0x32fd3cf5b4e49bb4,
+ 0xb7118682dbb66a77, 0x3fbc8c33221dc2a1,
+ 0xe4d5e82392a40515, 0xfabaf3feaa5334a,
+ 0x8f05b1163ba6832d, 0x29cb4d87f2a7400e,
+ 0xb2c71d5bca9023f8, 0x743e20e9ef511012,
+ 0xdf78e4b2bd342cf6, 0x914da9246b255416,
+ 0x8bab8eefb6409c1a, 0x1ad089b6c2f7548e,
+ 0xae9672aba3d0c320, 0xa184ac2473b529b1,
+ 0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741e,
+ 0x8865899617fb1871, 0x7e2fa67c7a658892,
+ 0xaa7eebfb9df9de8d, 0xddbb901b98feeab7,
+ 0xd51ea6fa85785631, 0x552a74227f3ea565,
+ 0x8533285c936b35de, 0xd53a88958f87275f,
+ 0xa67ff273b8460356, 0x8a892abaf368f137,
+ 0xd01fef10a657842c, 0x2d2b7569b0432d85,
+ 0x8213f56a67f6b29b, 0x9c3b29620e29fc73,
+ 0xa298f2c501f45f42, 0x8349f3ba91b47b8f,
+ 0xcb3f2f7642717713, 0x241c70a936219a73,
+ 0xfe0efb53d30dd4d7, 0xed238cd383aa0110,
+ 0x9ec95d1463e8a506, 0xf4363804324a40aa,
+ 0xc67bb4597ce2ce48, 0xb143c6053edcd0d5,
+ 0xf81aa16fdc1b81da, 0xdd94b7868e94050a,
+ 0x9b10a4e5e9913128, 0xca7cf2b4191c8326,
+ 0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f0,
+ 0xf24a01a73cf2dccf, 0xbc633b39673c8cec,
+ 0x976e41088617ca01, 0xd5be0503e085d813,
+ 0xbd49d14aa79dbc82, 0x4b2d8644d8a74e18,
+ 0xec9c459d51852ba2, 0xddf8e7d60ed1219e,
+ 0x93e1ab8252f33b45, 0xcabb90e5c942b503,
+ 0xb8da1662e7b00a17, 0x3d6a751f3b936243,
+ 0xe7109bfba19c0c9d, 0xcc512670a783ad4,
+ 0x906a617d450187e2, 0x27fb2b80668b24c5,
+ 0xb484f9dc9641e9da, 0xb1f9f660802dedf6,
+ 0xe1a63853bbd26451, 0x5e7873f8a0396973,
+ 0x8d07e33455637eb2, 0xdb0b487b6423e1e8,
+ 0xb049dc016abc5e5f, 0x91ce1a9a3d2cda62,
+ 0xdc5c5301c56b75f7, 0x7641a140cc7810fb,
+ 0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9d,
+ 0xac2820d9623bf429, 0x546345fa9fbdcd44,
+ 0xd732290fbacaf133, 0xa97c177947ad4095,
+ 0x867f59a9d4bed6c0, 0x49ed8eabcccc485d,
+ 0xa81f301449ee8c70, 0x5c68f256bfff5a74,
+ 0xd226fc195c6a2f8c, 0x73832eec6fff3111,
+ 0x83585d8fd9c25db7, 0xc831fd53c5ff7eab,
+ 0xa42e74f3d032f525, 0xba3e7ca8b77f5e55,
+ 0xcd3a1230c43fb26f, 0x28ce1bd2e55f35eb,
+ 0x80444b5e7aa7cf85, 0x7980d163cf5b81b3,
+ 0xa0555e361951c366, 0xd7e105bcc332621f,
+ 0xc86ab5c39fa63440, 0x8dd9472bf3fefaa7,
+ 0xfa856334878fc150, 0xb14f98f6f0feb951,
+ 0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d3,
+ 0xc3b8358109e84f07, 0xa862f80ec4700c8,
+ 0xf4a642e14c6262c8, 0xcd27bb612758c0fa,
+ 0x98e7e9cccfbd7dbd, 0x8038d51cb897789c,
+ 0xbf21e44003acdd2c, 0xe0470a63e6bd56c3,
+ 0xeeea5d5004981478, 0x1858ccfce06cac74,
+ 0x95527a5202df0ccb, 0xf37801e0c43ebc8,
+ 0xbaa718e68396cffd, 0xd30560258f54e6ba,
+ 0xe950df20247c83fd, 0x47c6b82ef32a2069,
+ 0x91d28b7416cdd27e, 0x4cdc331d57fa5441,
+ 0xb6472e511c81471d, 0xe0133fe4adf8e952,
+ 0xe3d8f9e563a198e5, 0x58180fddd97723a6,
+ 0x8e679c2f5e44ff8f, 0x570f09eaa7ea7648,
+ };
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <class unused>
+constexpr uint64_t
+ powers_template<unused>::power_of_five_128[number_of_entries];
+
+#endif
+
+using powers = powers_template<>;
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+#define SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+
+#include <cfloat>
+#include <cinttypes>
+#include <cmath>
+#include <cstdint>
+#include <cstdlib>
+#include <cstring>
+
+namespace simdjson_fast_float {
+
+// This will compute or rather approximate w * 5**q and return a pair of 64-bit
+// words approximating the result, with the "high" part corresponding to the
+// most significant bits and the low part corresponding to the least significant
+// bits.
+//
+template <int bit_precision>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+compute_product_approximation(int64_t q, uint64_t w) {
+ int const index = 2 * int(q - powers::smallest_power_of_five);
+ // For small values of q, e.g., q in [0,27], the answer is always exact
+ // because The line value128 firstproduct = full_multiplication(w,
+ // power_of_five_128[index]); gives the exact answer.
+ value128 firstproduct =
+ full_multiplication(w, powers::power_of_five_128[index]);
+ static_assert((bit_precision >= 0) && (bit_precision <= 64),
+ " precision should be in (0,64]");
+ constexpr uint64_t precision_mask =
+ (bit_precision < 64) ? (uint64_t(0xFFFFFFFFFFFFFFFF) >> bit_precision)
+ : uint64_t(0xFFFFFFFFFFFFFFFF);
+ if ((firstproduct.high & precision_mask) ==
+ precision_mask) { // could further guard with (lower + w < lower)
+ // regarding the second product, we only need secondproduct.high, but our
+ // expectation is that the compiler will optimize this extra work away if
+ // needed.
+ value128 secondproduct =
+ full_multiplication(w, powers::power_of_five_128[index + 1]);
+ firstproduct.low += secondproduct.high;
+ if (secondproduct.high > firstproduct.low) {
+ firstproduct.high++;
+ }
+ }
+ return firstproduct;
+}
+
+namespace detail {
+/**
+ * For q in (0,350), we have that
+ * f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ * floor(p) + q
+ * where
+ * p = log(5**q)/log(2) = q * log(5)/log(2)
+ *
+ * For negative values of q in (-400,0), we have that
+ * f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ * -ceil(p) + q
+ * where
+ * p = log(5**-q)/log(2) = -q * log(5)/log(2)
+ */
+constexpr simdjson_fastfloat_really_inline int32_t power(int32_t q) noexcept {
+ return (((152170 + 65536) * q) >> 16) + 63;
+}
+} // namespace detail
+
+// create an adjusted mantissa, biased by the invalid power2
+// for significant digits already multiplied by 10 ** q.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 adjusted_mantissa
+compute_error_scaled(int64_t q, uint64_t w, int lz) noexcept {
+ int hilz = int(w >> 63) ^ 1;
+ adjusted_mantissa answer;
+ answer.mantissa = w << hilz;
+ int bias = binary::mantissa_explicit_bits() - binary::minimum_exponent();
+ answer.power2 = int32_t(detail::power(int32_t(q)) + bias - hilz - lz - 62 +
+ invalid_am_bias);
+ return answer;
+}
+
+// w * 10 ** q, without rounding the representation up.
+// the power2 in the exponent will be adjusted by invalid_am_bias.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_error(int64_t q, uint64_t w) noexcept {
+ int lz = leading_zeroes(w);
+ w <<= lz;
+ value128 product =
+ compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+ return compute_error_scaled<binary>(q, product.high, lz);
+}
+
+// Computers w * 10 ** q.
+// The returned value should be a valid number that simply needs to be
+// packed. However, in some very rare cases, the computation will fail. In such
+// cases, we return an adjusted_mantissa with a negative power of 2: the caller
+// should recompute in such cases.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_float(int64_t q, uint64_t w) noexcept {
+ adjusted_mantissa answer;
+ if ((w == 0) || (q < binary::smallest_power_of_ten())) {
+ answer.power2 = 0;
+ answer.mantissa = 0;
+ // result should be zero
+ return answer;
+ }
+ if (q > binary::largest_power_of_ten()) {
+ // we want to get infinity:
+ answer.power2 = binary::infinite_power();
+ answer.mantissa = 0;
+ return answer;
+ }
+ // At this point in time q is in [powers::smallest_power_of_five,
+ // powers::largest_power_of_five].
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(w);
+ w <<= lz;
+
+ // The required precision is binary::mantissa_explicit_bits() + 3 because
+ // 1. We need the implicit bit
+ // 2. We need an extra bit for rounding purposes
+ // 3. We might lose a bit due to the "upperbit" routine (result too small,
+ // requiring a shift)
+
+ value128 product =
+ compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+ // The computed 'product' is always sufficient.
+ // Mathematical proof:
+ // Noble Mushtak and Daniel Lemire, Fast Number Parsing Without Fallback (to
+ // appear) See script/mushtak_lemire.py
+
+ // The "compute_product_approximation" function can be slightly slower than a
+ // branchless approach: value128 product = compute_product(q, w); but in
+ // practice, we can win big with the compute_product_approximation if its
+ // additional branch is easily predicted. Which is best is data specific.
+ int upperbit = int(product.high >> 63);
+ int shift = upperbit + 64 - binary::mantissa_explicit_bits() - 3;
+
+ answer.mantissa = product.high >> shift;
+
+ answer.power2 = int32_t(detail::power(int32_t(q)) + upperbit - lz -
+ binary::minimum_exponent());
+ if (answer.power2 <= 0) { // we have a subnormal?
+ // Here have that answer.power2 <= 0 so -answer.power2 >= 0
+ if (-answer.power2 + 1 >=
+ 64) { // if we have more than 64 bits below the minimum exponent, you
+ // have a zero for sure.
+ answer.power2 = 0;
+ answer.mantissa = 0;
+ // result should be zero
+ return answer;
+ }
+ // next line is safe because -answer.power2 + 1 < 64
+ answer.mantissa >>= -answer.power2 + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0 in the 32-bit and
+ // and 64-bit case (with no more than 19 digits).
+ answer.mantissa += (answer.mantissa & 1); // round up
+ answer.mantissa >>= 1;
+ // There is a weird scenario where we don't have a subnormal but just.
+ // Suppose we start with 2.2250738585072013e-308, we end up
+ // with 0x3fffffffffffff x 2^-1023-53 which is technically subnormal
+ // whereas 0x40000000000000 x 2^-1023-53 is normal. Now, we need to round
+ // up 0x3fffffffffffff x 2^-1023-53 and once we do, we are no longer
+ // subnormal, but we can only know this after rounding.
+ // So we only declare a subnormal if we are smaller than the threshold.
+ answer.power2 =
+ (answer.mantissa < (uint64_t(1) << binary::mantissa_explicit_bits()))
+ ? 0
+ : 1;
+ return answer;
+ }
+
+ // usually, we round *up*, but if we fall right in between and and we have an
+ // even basis, we need to round down
+ // We are only concerned with the cases where 5**q fits in single 64-bit word.
+ if ((product.low <= 1) && (q >= binary::min_exponent_round_to_even()) &&
+ (q <= binary::max_exponent_round_to_even()) &&
+ ((answer.mantissa & 3) == 1)) { // we may fall between two floats!
+ // To be in-between two floats we need that in doing
+ // answer.mantissa = product.high >> (upperbit + 64 -
+ // binary::mantissa_explicit_bits() - 3);
+ // ... we dropped out only zeroes. But if this happened, then we can go
+ // back!!!
+ if ((answer.mantissa << shift) == product.high) {
+ answer.mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ answer.mantissa += (answer.mantissa & 1); // round up
+ answer.mantissa >>= 1;
+ if (answer.mantissa >= (uint64_t(2) << binary::mantissa_explicit_bits())) {
+ answer.mantissa = (uint64_t(1) << binary::mantissa_explicit_bits());
+ answer.power2++; // undo previous addition
+ }
+
+ answer.mantissa &= ~(uint64_t(1) << binary::mantissa_explicit_bits());
+ if (answer.power2 >= binary::infinite_power()) { // infinity
+ answer.power2 = binary::infinite_power();
+ answer.mantissa = 0;
+ }
+ return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_BIGINT_H
+#define SIMDJSON_FASTFLOAT_BIGINT_H
+
+#include <algorithm>
+#include <cstdint>
+#include <climits>
+#include <cstring>
+
+
+namespace simdjson_fast_float {
+
+// the limb width: we want efficient multiplication of double the bits in
+// limb, or for 64-bit limbs, at least 64-bit multiplication where we can
+// extract the high and low parts efficiently. this is every 64-bit
+// architecture except for sparc, which emulates 128-bit multiplication.
+// we might have platforms where `CHAR_BIT` is not 8, so let's avoid
+// doing `8 * sizeof(limb)`.
+#if defined(SIMDJSON_FASTFLOAT_64BIT) && !defined(__sparc)
+#define SIMDJSON_FASTFLOAT_64BIT_LIMB 1
+typedef uint64_t limb;
+constexpr size_t limb_bits = 64;
+#else
+#define SIMDJSON_FASTFLOAT_32BIT_LIMB
+typedef uint32_t limb;
+constexpr size_t limb_bits = 32;
+#endif
+
+typedef span<limb> limb_span;
+
+// number of bits in a bigint. this needs to be at least the number
+// of bits required to store the largest bigint, which is
+// `log2(10**(digits + max_exp))`, or `log2(10**(767 + 342))`, or
+// ~3600 bits, so we round to 4000.
+constexpr size_t bigint_bits = 4000;
+constexpr size_t bigint_limbs = bigint_bits / limb_bits;
+
+// vector-like type that is allocated on the stack. the entire
+// buffer is pre-allocated, and only the length changes.
+template <uint16_t size> struct stackvec {
+ limb data[size];
+ // we never need more than 150 limbs
+ uint16_t length{0};
+
+ stackvec() = default;
+ stackvec(stackvec const &) = delete;
+ stackvec &operator=(stackvec const &) = delete;
+ stackvec(stackvec &&) = delete;
+ stackvec &operator=(stackvec &&other) = delete;
+
+ // create stack vector from existing limb span.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 stackvec(limb_span s) {
+ SIMDJSON_FASTFLOAT_ASSERT(try_extend(s));
+ }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 limb &operator[](size_t index) noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return data[index];
+ }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &operator[](size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return data[index];
+ }
+
+ // index from the end of the container
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &rindex(size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ size_t rindex = length - index - 1;
+ return data[rindex];
+ }
+
+ // set the length, without bounds checking.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void set_len(size_t len) noexcept {
+ length = uint16_t(len);
+ }
+
+ constexpr size_t len() const noexcept { return length; }
+
+ constexpr bool is_empty() const noexcept { return length == 0; }
+
+ constexpr size_t capacity() const noexcept { return size; }
+
+ // append item to vector, without bounds checking
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void push_unchecked(limb value) noexcept {
+ data[length] = value;
+ length++;
+ }
+
+ // append item to vector, returning if item was added
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 bool try_push(limb value) noexcept {
+ if (len() < capacity()) {
+ push_unchecked(value);
+ return true;
+ } else {
+ return false;
+ }
+ }
+
+ // add items to the vector, from a span, without bounds checking
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 void extend_unchecked(limb_span s) noexcept {
+ limb *ptr = data + length;
+ std::copy_n(s.ptr, s.len(), ptr);
+ set_len(len() + s.len());
+ }
+
+ // try to add items to the vector, returning if items were added
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_extend(limb_span s) noexcept {
+ if (len() + s.len() <= capacity()) {
+ extend_unchecked(s);
+ return true;
+ } else {
+ return false;
+ }
+ }
+
+ // resize the vector, without bounds checking
+ // if the new size is longer than the vector, assign value to each
+ // appended item.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20
+ void resize_unchecked(size_t new_len, limb value) noexcept {
+ if (new_len > len()) {
+ size_t count = new_len - len();
+ limb *first = data + len();
+ limb *last = first + count;
+ ::std::fill(first, last, value);
+ set_len(new_len);
+ } else {
+ set_len(new_len);
+ }
+ }
+
+ // try to resize the vector, returning if the vector was resized.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_resize(size_t new_len, limb value) noexcept {
+ if (new_len > capacity()) {
+ return false;
+ } else {
+ resize_unchecked(new_len, value);
+ return true;
+ }
+ }
+
+ // check if any limbs are non-zero after the given index.
+ // this needs to be done in reverse order, since the index
+ // is relative to the most significant limbs.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 bool nonzero(size_t index) const noexcept {
+ while (index < len()) {
+ if (rindex(index) != 0) {
+ return true;
+ }
+ index++;
+ }
+ return false;
+ }
+
+ // normalize the big integer, so most-significant zero limbs are removed.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void normalize() noexcept {
+ while (len() > 0 && rindex(0) == 0) {
+ length--;
+ }
+ }
+};
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+empty_hi64(bool &truncated) noexcept {
+ truncated = false;
+ return 0;
+}
- std::memmove(buf + (2 + static_cast<size_t>(-n)), buf,
- static_cast<size_t>(k));
- buf[0] = '0';
- buf[1] = '.';
- std::memset(buf + 2, '0', static_cast<size_t>(-n));
- return buf + (2U + static_cast<size_t>(-n) + static_cast<size_t>(k));
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, bool &truncated) noexcept {
+ truncated = false;
+ int shl = leading_zeroes(r0);
+ return r0 << shl;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, uint64_t r1, bool &truncated) noexcept {
+ int shl = leading_zeroes(r0);
+ if (shl == 0) {
+ truncated = r1 != 0;
+ return r0;
+ } else {
+ int shr = 64 - shl;
+ truncated = (r1 << shl) != 0;
+ return (r0 << shl) | (r1 >> shr);
}
+}
- if (k == 1) {
- // dE+123
- // len <= 1 + 5
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, bool &truncated) noexcept {
+ return uint64_hi64(r0, truncated);
+}
- buf += 1;
- } else {
- // d.igitsE+123
- // len <= max_digits10 + 1 + 5
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, bool &truncated) noexcept {
+ uint64_t x0 = r0;
+ uint64_t x1 = r1;
+ return uint64_hi64((x0 << 32) | x1, truncated);
+}
- std::memmove(buf + 2, buf + 1, static_cast<size_t>(k) - 1);
- buf[1] = '.';
- buf += 1 + static_cast<size_t>(k);
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, uint32_t r2, bool &truncated) noexcept {
+ uint64_t x0 = r0;
+ uint64_t x1 = r1;
+ uint64_t x2 = r2;
+ return uint64_hi64(x0, (x1 << 32) | x2, truncated);
+}
+
+// add two small integers, checking for overflow.
+// we want an efficient operation. for msvc, where
+// we don't have built-in intrinsics, this is still
+// pretty fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_add(limb x, limb y, bool &overflow) noexcept {
+ limb z;
+// gcc and clang
+#if defined(__has_builtin)
+#if __has_builtin(__builtin_add_overflow)
+ if (!cpp20_and_in_constexpr()) {
+ overflow = __builtin_add_overflow(x, y, &z);
+ return z;
}
+#endif
+#endif
- *buf++ = 'e';
- return append_exponent(buf, n - 1);
+ // generic, this still optimizes correctly on MSVC.
+ z = x + y;
+ overflow = z < x;
+ return z;
+}
+
+// multiply two small integers, getting both the high and low bits.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_mul(limb x, limb y, limb &carry) noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+#if defined(__SIZEOF_INT128__)
+ // GCC and clang both define it as an extension.
+ __uint128_t z = __uint128_t(x) * __uint128_t(y) + __uint128_t(carry);
+ carry = limb(z >> limb_bits);
+ return limb(z);
+#else
+ // fallback, no native 128-bit integer multiplication with carry.
+ // on msvc, this optimizes identically, somehow.
+ value128 z = full_multiplication(x, y);
+ bool overflow;
+ z.low = scalar_add(z.low, carry, overflow);
+ z.high += uint64_t(overflow); // cannot overflow
+ carry = z.high;
+ return z.low;
+#endif
+#else
+ uint64_t z = uint64_t(x) * uint64_t(y) + uint64_t(carry);
+ carry = limb(z >> limb_bits);
+ return limb(z);
+#endif
}
-} // namespace dtoa_impl
+// add scalar value to bigint starting from offset.
+// used in grade school multiplication
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_add_from(stackvec<size> &vec, limb y,
+ size_t start) noexcept {
+ size_t index = start;
+ limb carry = y;
+ bool overflow;
+ while (carry != 0 && index < vec.len()) {
+ vec[index] = scalar_add(vec[index], carry, overflow);
+ carry = limb(overflow);
+ index += 1;
+ }
+ if (carry != 0) {
+ SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
+ }
+ return true;
+}
-/*!
-The format of the resulting decimal representation is similar to printf's %g
-format. Returns an iterator pointing past-the-end of the decimal representation.
-@note The input number must be finite, i.e. NaN's and Inf's are not supported.
-@note The buffer must be large enough.
-@note The result is NOT null-terminated.
-*/
-char *to_chars(char *first, const char *last, double value) {
- static_cast<void>(last); // maybe unused - fix warning
- bool negative = std::signbit(value);
- if (negative) {
- value = -value;
- *first++ = '-';
+// add scalar value to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+small_add(stackvec<size> &vec, limb y) noexcept {
+ return small_add_from(vec, y, 0);
+}
+
+// multiply bigint by scalar value.
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_mul(stackvec<size> &vec,
+ limb y) noexcept {
+ limb carry = 0;
+ for (size_t index = 0; index < vec.len(); index++) {
+ vec[index] = scalar_mul(vec[index], y, carry);
+ }
+ if (carry != 0) {
+ SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
}
+ return true;
+}
- if (value == 0) // +-0
- {
- *first++ = '0';
- // Make it look like a floating-point number (#362, #378)
- *first++ = '.';
- *first++ = '0';
- return first;
+// add bigint to bigint starting from index.
+// used in grade school multiplication
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_add_from(stackvec<size> &x, limb_span y,
+ size_t start) noexcept {
+ // the effective x buffer is from `xstart..x.len()`, so exit early
+ // if we can't get that current range.
+ if (x.len() < start || y.len() > x.len() - start) {
+ SIMDJSON_FASTFLOAT_TRY(x.try_resize(y.len() + start, 0));
}
- // Compute v = buffer * 10^decimal_exponent.
- // The decimal digits are stored in the buffer, which needs to be interpreted
- // as an unsigned decimal integer.
- // len is the length of the buffer, i.e. the number of decimal digits.
- int len = 0;
- int decimal_exponent = 0;
- dtoa_impl::grisu2(first, len, decimal_exponent, value);
- // Format the buffer like printf("%.*g", prec, value)
- constexpr int kMinExp = -4;
- constexpr int kMaxExp = std::numeric_limits<double>::digits10;
- return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp,
- kMaxExp);
+ bool carry = false;
+ for (size_t index = 0; index < y.len(); index++) {
+ limb xi = x[index + start];
+ limb yi = y[index];
+ bool c1 = false;
+ bool c2 = false;
+ xi = scalar_add(xi, yi, c1);
+ if (carry) {
+ xi = scalar_add(xi, 1, c2);
+ }
+ x[index + start] = xi;
+ carry = c1 | c2;
+ }
+
+ // handle overflow
+ if (carry) {
+ SIMDJSON_FASTFLOAT_TRY(small_add_from(x, 1, y.len() + start));
+ }
+ return true;
}
-} // namespace internal
-} // namespace simdjson
-#endif // SIMDJSON_SRC_TO_CHARS_CPP
-/* end file to_chars.cpp */
-/* including from_chars.cpp: #include <from_chars.cpp> */
-/* begin file from_chars.cpp */
-#ifndef SIMDJSON_SRC_FROM_CHARS_CPP
-#define SIMDJSON_SRC_FROM_CHARS_CPP
+// add bigint to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+large_add_from(stackvec<size> &x, limb_span y) noexcept {
+ return large_add_from(x, y, 0);
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool long_mul(stackvec<size> &x, limb_span y) noexcept {
+ limb_span xs = limb_span(x.data, x.len());
+ stackvec<size> z(xs);
+ limb_span zs = limb_span(z.data, z.len());
+
+ if (y.len() != 0) {
+ limb y0 = y[0];
+ SIMDJSON_FASTFLOAT_TRY(small_mul(x, y0));
+ for (size_t index = 1; index < y.len(); index++) {
+ limb yi = y[index];
+ stackvec<size> zi;
+ if (yi != 0) {
+ // re-use the same buffer throughout
+ zi.set_len(0);
+ SIMDJSON_FASTFLOAT_TRY(zi.try_extend(zs));
+ SIMDJSON_FASTFLOAT_TRY(small_mul(zi, yi));
+ limb_span zis = limb_span(zi.data, zi.len());
+ SIMDJSON_FASTFLOAT_TRY(large_add_from(x, zis, index));
+ }
+ }
+ }
-/* skipped duplicate #include <base.h> */
+ x.normalize();
+ return true;
+}
-#include <cstdint>
-#include <cstring>
-#include <limits>
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_mul(stackvec<size> &x, limb_span y) noexcept {
+ if (y.len() == 1) {
+ SIMDJSON_FASTFLOAT_TRY(small_mul(x, y[0]));
+ } else {
+ SIMDJSON_FASTFLOAT_TRY(long_mul(x, y));
+ }
+ return true;
+}
-namespace simdjson {
-namespace internal {
+template <typename = void> struct pow5_tables {
+ static constexpr uint32_t large_step = 135;
+ static constexpr uint64_t small_power_of_5[] = {
+ 1UL,
+ 5UL,
+ 25UL,
+ 125UL,
+ 625UL,
+ 3125UL,
+ 15625UL,
+ 78125UL,
+ 390625UL,
+ 1953125UL,
+ 9765625UL,
+ 48828125UL,
+ 244140625UL,
+ 1220703125UL,
+ 6103515625UL,
+ 30517578125UL,
+ 152587890625UL,
+ 762939453125UL,
+ 3814697265625UL,
+ 19073486328125UL,
+ 95367431640625UL,
+ 476837158203125UL,
+ 2384185791015625UL,
+ 11920928955078125UL,
+ 59604644775390625UL,
+ 298023223876953125UL,
+ 1490116119384765625UL,
+ 7450580596923828125UL,
+ };
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ constexpr static limb large_power_of_5[] = {
+ 1414648277510068013UL, 9180637584431281687UL, 4539964771860779200UL,
+ 10482974169319127550UL, 198276706040285095UL};
+#else
+ constexpr static limb large_power_of_5[] = {
+ 4279965485U, 329373468U, 4020270615U, 2137533757U, 4287402176U,
+ 1057042919U, 1071430142U, 2440757623U, 381945767U, 46164893U};
+#endif
+};
-/**
- * The code in the internal::from_chars function is meant to handle the floating-point number parsing
- * when we have more than 19 digits in the decimal mantissa. This should only be seen
- * in adversarial scenarios: we do not expect production systems to even produce
- * such floating-point numbers.
- *
- * The parser is based on work by Nigel Tao (at https://github.com/google/wuffs/)
- * who credits Ken Thompson for the design (via a reference to the Go source
- * code). See
- * https://github.com/google/wuffs/blob/aa46859ea40c72516deffa1b146121952d6dfd3b/internal/cgen/base/floatconv-submodule-data.c
- * https://github.com/google/wuffs/blob/46cd8105f47ca07ae2ba8e6a7818ef9c0df6c152/internal/cgen/base/floatconv-submodule-code.c
- * It is probably not very fast but it is a fallback that should almost never be
- * called in real life. Google Wuffs is published under APL 2.0.
- **/
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
-namespace {
-constexpr uint32_t max_digits = 768;
-constexpr int32_t decimal_point_range = 2047;
-} // namespace
+template <typename T> constexpr uint32_t pow5_tables<T>::large_step;
-struct adjusted_mantissa {
- uint64_t mantissa;
- int power2;
- adjusted_mantissa() : mantissa(0), power2(0) {}
-};
+template <typename T> constexpr uint64_t pow5_tables<T>::small_power_of_5[];
-struct decimal {
- uint32_t num_digits;
- int32_t decimal_point;
- bool negative;
- bool truncated;
- uint8_t digits[max_digits];
-};
+template <typename T> constexpr limb pow5_tables<T>::large_power_of_5[];
-template <typename T> struct binary_format {
- static constexpr int mantissa_explicit_bits();
- static constexpr int minimum_exponent();
- static constexpr int infinite_power();
- static constexpr int sign_index();
-};
+#endif
-template <> constexpr int binary_format<double>::mantissa_explicit_bits() {
- return 52;
-}
+// big integer type. implements a small subset of big integer
+// arithmetic, using simple algorithms since asymptotically
+// faster algorithms are slower for a small number of limbs.
+// all operations assume the big-integer is normalized.
+struct bigint : pow5_tables<> {
+ // storage of the limbs, in little-endian order.
+ stackvec<bigint_limbs> vec;
-template <> constexpr int binary_format<double>::minimum_exponent() {
- return -1023;
-}
-template <> constexpr int binary_format<double>::infinite_power() {
- return 0x7FF;
-}
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint() : vec() {}
-template <> constexpr int binary_format<double>::sign_index() { return 63; }
+ bigint(bigint const &) = delete;
+ bigint &operator=(bigint const &) = delete;
+ bigint(bigint &&) = delete;
+ bigint &operator=(bigint &&other) = delete;
-bool is_integer(char c) noexcept { return (c >= '0' && c <= '9'); }
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint(uint64_t value) : vec() {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ vec.push_unchecked(value);
+#else
+ vec.push_unchecked(uint32_t(value));
+ vec.push_unchecked(uint32_t(value >> 32));
+#endif
+ vec.normalize();
+ }
-// This should always succeed since it follows a call to parse_number.
-decimal parse_decimal(const char *&p) noexcept {
- decimal answer;
- answer.num_digits = 0;
- answer.decimal_point = 0;
- answer.truncated = false;
- answer.negative = (*p == '-');
- if ((*p == '-') || (*p == '+')) {
- ++p;
+ // get the high 64 bits from the vector, and if bits were truncated.
+ // this is to get the significant digits for the float.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t hi64(bool &truncated) const noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ if (vec.len() == 0) {
+ return empty_hi64(truncated);
+ } else if (vec.len() == 1) {
+ return uint64_hi64(vec.rindex(0), truncated);
+ } else {
+ uint64_t result = uint64_hi64(vec.rindex(0), vec.rindex(1), truncated);
+ truncated |= vec.nonzero(2);
+ return result;
+ }
+#else
+ if (vec.len() == 0) {
+ return empty_hi64(truncated);
+ } else if (vec.len() == 1) {
+ return uint32_hi64(vec.rindex(0), truncated);
+ } else if (vec.len() == 2) {
+ return uint32_hi64(vec.rindex(0), vec.rindex(1), truncated);
+ } else {
+ uint64_t result =
+ uint32_hi64(vec.rindex(0), vec.rindex(1), vec.rindex(2), truncated);
+ truncated |= vec.nonzero(3);
+ return result;
+ }
+#endif
}
- while (*p == '0') {
- ++p;
+ // compare two big integers, returning the large value.
+ // assumes both are normalized. if the return value is
+ // negative, other is larger, if the return value is
+ // positive, this is larger, otherwise they are equal.
+ // the limbs are stored in little-endian order, so we
+ // must compare the limbs in ever order.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int compare(bigint const &other) const noexcept {
+ if (vec.len() > other.vec.len()) {
+ return 1;
+ } else if (vec.len() < other.vec.len()) {
+ return -1;
+ } else {
+ for (size_t index = vec.len(); index > 0; index--) {
+ limb xi = vec[index - 1];
+ limb yi = other.vec[index - 1];
+ if (xi > yi) {
+ return 1;
+ } else if (xi < yi) {
+ return -1;
+ }
+ }
+ return 0;
+ }
}
- while (is_integer(*p)) {
- if (answer.num_digits < max_digits) {
- answer.digits[answer.num_digits] = uint8_t(*p - '0');
+
+ // shift left each limb n bits, carrying over to the new limb
+ // returns true if we were able to shift all the digits.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_bits(size_t n) noexcept {
+ // Internally, for each item, we shift left by n, and add the previous
+ // right shifted limb-bits.
+ // For example, we transform (for u8) shifted left 2, to:
+ // b10100100 b01000010
+ // b10 b10010001 b00001000
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n < sizeof(limb) * 8);
+
+ size_t shl = n;
+ size_t shr = limb_bits - shl;
+ limb prev = 0;
+ for (size_t index = 0; index < vec.len(); index++) {
+ limb xi = vec[index];
+ vec[index] = (xi << shl) | (prev >> shr);
+ prev = xi;
}
- answer.num_digits++;
- ++p;
+
+ limb carry = prev >> shr;
+ if (carry != 0) {
+ return vec.try_push(carry);
+ }
+ return true;
}
- if (*p == '.') {
- ++p;
- const char *first_after_period = p;
- // if we have not yet encountered a zero, we have to skip it as well
- if (answer.num_digits == 0) {
- // skip zeros
- while (*p == '0') {
- ++p;
- }
+
+ // move the limbs left by `n` limbs.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_limbs(size_t n) noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+ if (n + vec.len() > vec.capacity()) {
+ return false;
+ } else if (!vec.is_empty()) {
+ // move limbs
+ limb *dst = vec.data + n;
+ limb const *src = vec.data;
+ std::copy_backward(src, src + vec.len(), dst + vec.len());
+ // fill in empty limbs
+ limb *first = vec.data;
+ limb *last = first + n;
+ ::std::fill(first, last, 0);
+ vec.set_len(n + vec.len());
+ return true;
+ } else {
+ return true;
}
- while (is_integer(*p)) {
- if (answer.num_digits < max_digits) {
- answer.digits[answer.num_digits] = uint8_t(*p - '0');
- }
- answer.num_digits++;
- ++p;
+ }
+
+ // move the limbs left by `n` bits.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl(size_t n) noexcept {
+ size_t rem = n % limb_bits;
+ size_t div = n / limb_bits;
+ if (rem != 0) {
+ SIMDJSON_FASTFLOAT_TRY(shl_bits(rem));
}
- answer.decimal_point = int32_t(first_after_period - p);
+ if (div != 0) {
+ SIMDJSON_FASTFLOAT_TRY(shl_limbs(div));
+ }
+ return true;
}
- if(answer.num_digits > 0) {
- const char *preverse = p - 1;
- int32_t trailing_zeros = 0;
- while ((*preverse == '0') || (*preverse == '.')) {
- if(*preverse == '0') { trailing_zeros++; };
- --preverse;
+
+ // get the number of leading zeros in the bigint.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int ctlz() const noexcept {
+ if (vec.is_empty()) {
+ return 0;
+ } else {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ return leading_zeroes(vec.rindex(0));
+#else
+ // no use defining a specialized leading_zeroes for a 32-bit type.
+ uint64_t r0 = vec.rindex(0);
+ return leading_zeroes(r0 << 32);
+#endif
}
- answer.decimal_point += int32_t(answer.num_digits);
- answer.num_digits -= uint32_t(trailing_zeros);
}
- if(answer.num_digits > max_digits ) {
- answer.num_digits = max_digits;
- answer.truncated = true;
+
+ // get the number of bits in the bigint.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int bit_length() const noexcept {
+ int lz = ctlz();
+ return int(limb_bits * vec.len()) - lz;
}
- if (('e' == *p) || ('E' == *p)) {
- ++p;
- bool neg_exp = false;
- if ('-' == *p) {
- neg_exp = true;
- ++p;
- } else if ('+' == *p) {
- ++p;
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool mul(limb y) noexcept { return small_mul(vec, y); }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool add(limb y) noexcept { return small_add(vec, y); }
+
+ // multiply as if by 2 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow2(uint32_t exp) noexcept { return shl(exp); }
+
+ // multiply as if by 5 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow5(uint32_t exp) noexcept {
+ // multiply by a power of 5
+ size_t large_length = sizeof(large_power_of_5) / sizeof(limb);
+ limb_span large = limb_span(large_power_of_5, large_length);
+ while (exp >= large_step) {
+ SIMDJSON_FASTFLOAT_TRY(large_mul(vec, large));
+ exp -= large_step;
}
- int32_t exp_number = 0; // exponential part
- while (is_integer(*p)) {
- uint8_t digit = uint8_t(*p - '0');
- if (exp_number < 0x10000) {
- exp_number = 10 * exp_number + digit;
- }
- ++p;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ uint32_t small_step = 27;
+ limb max_native = 7450580596923828125UL;
+#else
+ uint32_t small_step = 13;
+ limb max_native = 1220703125U;
+#endif
+ while (exp >= small_step) {
+ SIMDJSON_FASTFLOAT_TRY(small_mul(vec, max_native));
+ exp -= small_step;
+ }
+ if (exp != 0) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ // This is similar to https://github.com/llvm/llvm-project/issues/47746,
+ // except the workaround described there don't work here
+ SIMDJSON_FASTFLOAT_TRY(small_mul(vec, limb((static_cast<void>(small_power_of_5[0]),
+ small_power_of_5[exp]))));
}
- answer.decimal_point += (neg_exp ? -exp_number : exp_number);
+
+ return true;
}
- return answer;
+
+ // multiply as if by 10 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow10(uint32_t exp) noexcept {
+ SIMDJSON_FASTFLOAT_TRY(pow5(exp));
+ return pow2(exp);
+ }
+};
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+#define SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+
+
+namespace simdjson_fast_float {
+
+// 1e0 to 1e19
+constexpr static uint64_t powers_of_ten_uint64[] = {1UL,
+ 10UL,
+ 100UL,
+ 1000UL,
+ 10000UL,
+ 100000UL,
+ 1000000UL,
+ 10000000UL,
+ 100000000UL,
+ 1000000000UL,
+ 10000000000UL,
+ 100000000000UL,
+ 1000000000000UL,
+ 10000000000000UL,
+ 100000000000000UL,
+ 1000000000000000UL,
+ 10000000000000000UL,
+ 100000000000000000UL,
+ 1000000000000000000UL,
+ 10000000000000000000UL};
+
+// calculate the exponent, in scientific notation, of the number.
+// this algorithm is not even close to optimized, but it has no practical
+// effect on performance: in order to have a faster algorithm, we'd need
+// to slow down performance for faster algorithms, and this is still fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int32_t
+scientific_exponent(uint64_t mantissa, int32_t exponent) noexcept {
+ while (mantissa >= 10000) {
+ mantissa /= 10000;
+ exponent += 4;
+ }
+ while (mantissa >= 100) {
+ mantissa /= 100;
+ exponent += 2;
+ }
+ while (mantissa >= 10) {
+ mantissa /= 10;
+ exponent += 1;
+ }
+ return exponent;
+}
+
+// this converts a native floating-point number to an extended-precision float.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended(T value) noexcept {
+ using equiv_uint = equiv_uint_t<T>;
+ constexpr equiv_uint exponent_mask = binary_format<T>::exponent_mask();
+ constexpr equiv_uint mantissa_mask = binary_format<T>::mantissa_mask();
+ constexpr equiv_uint hidden_bit_mask = binary_format<T>::hidden_bit_mask();
+
+ adjusted_mantissa am;
+ int32_t bias = binary_format<T>::mantissa_explicit_bits() -
+ binary_format<T>::minimum_exponent();
+ equiv_uint bits;
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ bits = std::bit_cast<equiv_uint>(value);
+#else
+ ::memcpy(&bits, &value, sizeof(T));
+#endif
+ if ((bits & exponent_mask) == 0) {
+ // denormal
+ am.power2 = 1 - bias;
+ am.mantissa = bits & mantissa_mask;
+ } else {
+ // normal
+ am.power2 = int32_t((bits & exponent_mask) >>
+ binary_format<T>::mantissa_explicit_bits());
+ am.power2 -= bias;
+ am.mantissa = (bits & mantissa_mask) | hidden_bit_mask;
+ }
+
+ return am;
}
-// This should always succeed since it follows a call to parse_number.
-// Will not read at or beyond the "end" pointer.
-decimal parse_decimal(const char *&p, const char * end) noexcept {
- decimal answer;
- answer.num_digits = 0;
- answer.decimal_point = 0;
- answer.truncated = false;
- if(p == end) { return answer; } // should never happen
- answer.negative = (*p == '-');
- if ((*p == '-') || (*p == '+')) {
- ++p;
+// get the extended precision value of the halfway point between b and b+u.
+// we are given a native float that represents b, so we need to adjust it
+// halfway between b and b+u.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended_halfway(T value) noexcept {
+ adjusted_mantissa am = to_extended(value);
+ am.mantissa <<= 1;
+ am.mantissa += 1;
+ am.power2 -= 1;
+ return am;
+}
+
+// round an extended-precision float to the nearest machine float.
+template <typename T, typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void round(adjusted_mantissa &am,
+ callback cb) noexcept {
+ int32_t mantissa_shift = 64 - binary_format<T>::mantissa_explicit_bits() - 1;
+ if (-am.power2 >= mantissa_shift) {
+ // have a denormal float
+ int32_t shift = -am.power2 + 1;
+ cb(am, (shift < 64 ? shift : 64));
+ // check for round-up: if rounding-nearest carried us to the hidden bit.
+ am.power2 = (am.mantissa <
+ (uint64_t(1) << binary_format<T>::mantissa_explicit_bits()))
+ ? 0
+ : 1;
+ return;
}
- while ((p != end) && (*p == '0')) {
- ++p;
+ // have a normal float, use the default shift.
+ cb(am, mantissa_shift);
+
+ // check for carry
+ if (am.mantissa >=
+ (uint64_t(2) << binary_format<T>::mantissa_explicit_bits())) {
+ am.mantissa = (uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+ am.power2++;
+ }
+
+ // check for infinite: we could have carried to an infinite power
+ am.mantissa &= ~(uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+ if (am.power2 >= binary_format<T>::infinite_power()) {
+ am.power2 = binary_format<T>::infinite_power();
+ am.mantissa = 0;
+ }
+}
+
+template <typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_nearest_tie_even(adjusted_mantissa &am, int32_t shift,
+ callback cb) noexcept {
+ uint64_t const mask = (shift == 64) ? UINT64_MAX : (uint64_t(1) << shift) - 1;
+ uint64_t const halfway = (shift == 0) ? 0 : uint64_t(1) << (shift - 1);
+ uint64_t truncated_bits = am.mantissa & mask;
+ bool is_above = truncated_bits > halfway;
+ bool is_halfway = truncated_bits == halfway;
+
+ // shift digits into position
+ if (shift == 64) {
+ am.mantissa = 0;
+ } else {
+ am.mantissa >>= shift;
+ }
+ am.power2 += shift;
+
+ bool is_odd = (am.mantissa & 1) == 1;
+ am.mantissa += uint64_t(cb(is_odd, is_halfway, is_above));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_down(adjusted_mantissa &am, int32_t shift) noexcept {
+ if (shift == 64) {
+ am.mantissa = 0;
+ } else {
+ am.mantissa >>= shift;
}
- while ((p != end) && is_integer(*p)) {
- if (answer.num_digits < max_digits) {
- answer.digits[answer.num_digits] = uint8_t(*p - '0');
+ am.power2 += shift;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+skip_zeros(UC const *&first, UC const *last) noexcept {
+ uint64_t val;
+ while (!cpp20_and_in_constexpr() &&
+ std::distance(first, last) >= int_cmp_len<UC>()) {
+ ::memcpy(&val, first, sizeof(uint64_t));
+ if (val != int_cmp_zeros<UC>()) {
+ break;
}
- answer.num_digits++;
- ++p;
+ first += int_cmp_len<UC>();
}
- if ((p != end) && (*p == '.')) {
- ++p;
- if(p == end) { return answer; } // should never happen
- const char *first_after_period = p;
- // if we have not yet encountered a zero, we have to skip it as well
- if (answer.num_digits == 0) {
- // skip zeros
- while (*p == '0') {
- ++p;
- }
+ while (first != last) {
+ if (*first != UC('0')) {
+ break;
}
- while ((p != end) && is_integer(*p)) {
- if (answer.num_digits < max_digits) {
- answer.digits[answer.num_digits] = uint8_t(*p - '0');
- }
- answer.num_digits++;
- ++p;
+ first++;
+ }
+}
+
+// determine if any non-zero digits were truncated.
+// all characters must be valid digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(UC const *first, UC const *last) noexcept {
+ // do 8-bit optimizations, can just compare to 8 literal 0s.
+ uint64_t val;
+ while (!cpp20_and_in_constexpr() &&
+ std::distance(first, last) >= int_cmp_len<UC>()) {
+ ::memcpy(&val, first, sizeof(uint64_t));
+ if (val != int_cmp_zeros<UC>()) {
+ return true;
}
- answer.decimal_point = int32_t(first_after_period - p);
+ first += int_cmp_len<UC>();
}
- if(answer.num_digits > 0) {
- const char *preverse = p - 1;
- int32_t trailing_zeros = 0;
- while ((*preverse == '0') || (*preverse == '.')) {
- if(*preverse == '0') { trailing_zeros++; };
- --preverse;
+ while (first != last) {
+ if (*first != UC('0')) {
+ return true;
}
- answer.decimal_point += int32_t(answer.num_digits);
- answer.num_digits -= uint32_t(trailing_zeros);
+ ++first;
}
- if(answer.num_digits > max_digits ) {
- answer.num_digits = max_digits;
- answer.truncated = true;
+ return false;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(span<UC const> s) noexcept {
+ return is_truncated(s.ptr, s.ptr + s.len());
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_eight_digits(UC const *&p, limb &value, size_t &counter,
+ size_t &count) noexcept {
+ value = value * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ counter += 8;
+ count += 8;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+parse_one_digit(UC const *&p, limb &value, size_t &counter,
+ size_t &count) noexcept {
+ value = value * 10 + limb(*p - UC('0'));
+ p++;
+ counter++;
+ count++;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+add_native(bigint &big, limb power, limb value) noexcept {
+ big.mul(power);
+ big.add(value);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+round_up_bigint(bigint &big, size_t &count) noexcept {
+ // need to round-up the digits, but need to avoid rounding
+ // ....9999 to ...10000, which could cause a false halfway point.
+ add_native(big, 10, 1);
+ count++;
+}
+
+// parse the significant digits into a big integer
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_mantissa(bigint &result, parsed_number_string_t<UC> &num,
+ size_t max_digits, size_t &digits) noexcept {
+ // try to minimize the number of big integer and scalar multiplication.
+ // therefore, try to parse 8 digits at a time, and multiply by the largest
+ // scalar value (9 or 19 digits) for each step.
+ size_t counter = 0;
+ digits = 0;
+ limb value = 0;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ size_t step = 19;
+#else
+ size_t step = 9;
+#endif
+
+ // process all integer digits.
+ UC const *p = num.integer.ptr;
+ UC const *pend = p + num.integer.len();
+ skip_zeros(p, pend);
+ // process all digits, in increments of step per loop
+ while (p != pend) {
+ while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+ (max_digits - digits >= 8)) {
+ parse_eight_digits(p, value, counter, digits);
+ }
+ while (counter < step && p != pend && digits < max_digits) {
+ parse_one_digit(p, value, counter, digits);
+ }
+ if (digits == max_digits) {
+ // add the temporary value, then check if we've truncated any digits
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ bool truncated = is_truncated(p, pend);
+ if (num.fraction.ptr != nullptr) {
+ truncated |= is_truncated(num.fraction);
+ }
+ if (truncated) {
+ round_up_bigint(result, digits);
+ }
+ return;
+ } else {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ counter = 0;
+ value = 0;
+ }
}
- if ((p != end) && (('e' == *p) || ('E' == *p))) {
- ++p;
- if(p == end) { return answer; } // should never happen
- bool neg_exp = false;
- if ('-' == *p) {
- neg_exp = true;
- ++p;
- } else if ('+' == *p) {
- ++p;
+
+ // add our fraction digits, if they're available.
+ if (num.fraction.ptr != nullptr) {
+ p = num.fraction.ptr;
+ pend = p + num.fraction.len();
+ if (digits == 0) {
+ skip_zeros(p, pend);
}
- int32_t exp_number = 0; // exponential part
- while ((p != end) && is_integer(*p)) {
- uint8_t digit = uint8_t(*p - '0');
- if (exp_number < 0x10000) {
- exp_number = 10 * exp_number + digit;
+ // process all digits, in increments of step per loop
+ while (p != pend) {
+ while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+ (max_digits - digits >= 8)) {
+ parse_eight_digits(p, value, counter, digits);
+ }
+ while (counter < step && p != pend && digits < max_digits) {
+ parse_one_digit(p, value, counter, digits);
+ }
+ if (digits == max_digits) {
+ // add the temporary value, then check if we've truncated any digits
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ bool truncated = is_truncated(p, pend);
+ if (truncated) {
+ round_up_bigint(result, digits);
+ }
+ return;
+ } else {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ counter = 0;
+ value = 0;
}
- ++p;
}
- answer.decimal_point += (neg_exp ? -exp_number : exp_number);
}
+
+ if (counter != 0) {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ }
+}
+
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+positive_digit_comp(bigint &bigmant, int32_t exponent) noexcept {
+ SIMDJSON_FASTFLOAT_ASSERT(bigmant.pow10(uint32_t(exponent)));
+ adjusted_mantissa answer;
+ bool truncated;
+ answer.mantissa = bigmant.hi64(truncated);
+ int bias = binary_format<T>::mantissa_explicit_bits() -
+ binary_format<T>::minimum_exponent();
+ answer.power2 = bigmant.bit_length() - 64 + bias;
+
+ round<T>(answer, [truncated](adjusted_mantissa &a, int32_t shift) {
+ round_nearest_tie_even(
+ a, shift,
+ [truncated](bool is_odd, bool is_halfway, bool is_above) -> bool {
+ return is_above || (is_halfway && truncated) ||
+ (is_odd && is_halfway);
+ });
+ });
+
return answer;
}
-namespace {
+// the scaling here is quite simple: we have, for the real digits `m * 10^e`,
+// and for the theoretical digits `n * 2^f`. Since `e` is always negative,
+// to scale them identically, we do `n * 2^f * 5^-f`, so we now have `m * 2^e`.
+// we then need to scale by `2^(f- e)`, and then the two significant digits
+// are of the same magnitude.
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa negative_digit_comp(
+ bigint &bigmant, adjusted_mantissa am, int32_t exponent) noexcept {
+ bigint &real_digits = bigmant;
+ int32_t real_exp = exponent;
+
+ // get the value of `b`, rounded down, and get a bigint representation of b+h
+ adjusted_mantissa am_b = am;
+ // gcc7 buf: use a lambda to remove the noexcept qualifier bug with
+ // -Wnoexcept-type.
+ round<T>(am_b,
+ [](adjusted_mantissa &a, int32_t shift) { round_down(a, shift); });
+ T b;
+ to_float(false, am_b, b);
+ adjusted_mantissa theor = to_extended_halfway(b);
+ bigint theor_digits(theor.mantissa);
+ int32_t theor_exp = theor.power2;
+
+ // scale real digits and theor digits to be same power.
+ int32_t pow2_exp = theor_exp - real_exp;
+ uint32_t pow5_exp = uint32_t(-real_exp);
+ if (pow5_exp != 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow5(pow5_exp));
+ }
+ if (pow2_exp > 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow2(uint32_t(pow2_exp)));
+ } else if (pow2_exp < 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(real_digits.pow2(uint32_t(-pow2_exp)));
+ }
+
+ // compare digits, and use it to direct rounding
+ int ord = real_digits.compare(theor_digits);
+ adjusted_mantissa answer = am;
+ round<T>(answer, [ord](adjusted_mantissa &a, int32_t shift) {
+ round_nearest_tie_even(
+ a, shift, [ord](bool is_odd, bool _, bool __) -> bool {
+ static_cast<void>(_); // not needed, since we've done our comparison
+ static_cast<void>(__); // not needed, since we've done our comparison
+ if (ord > 0) {
+ return true;
+ } else if (ord < 0) {
+ return false;
+ } else {
+ return is_odd;
+ }
+ });
+ });
+
+ return answer;
+}
-// remove all final zeroes
-inline void trim(decimal &h) {
- while ((h.num_digits > 0) && (h.digits[h.num_digits - 1] == 0)) {
- h.num_digits--;
+// parse the significant digits as a big integer to unambiguously round
+// the significant digits. here, we are trying to determine how to round
+// an extended float representation close to `b+h`, halfway between `b`
+// (the float rounded-down) and `b+u`, the next positive float. this
+// algorithm is always correct, and uses one of two approaches. when
+// the exponent is positive relative to the significant digits (such as
+// 1234), we create a big-integer representation, get the high 64-bits,
+// determine if any lower bits are truncated, and use that to direct
+// rounding. in case of a negative exponent relative to the significant
+// digits (such as 1.2345), we create a theoretical representation of
+// `b` as a big-integer type, scaled to the same binary exponent as
+// the actual digits. we then compare the big integer representations
+// of both, and use that to direct rounding.
+template <typename T, typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+digit_comp(parsed_number_string_t<UC> &num, adjusted_mantissa am) noexcept {
+ // remove the invalid exponent bias
+ am.power2 -= invalid_am_bias;
+
+ int32_t sci_exp =
+ scientific_exponent(num.mantissa, static_cast<int32_t>(num.exponent));
+ size_t max_digits = binary_format<T>::max_digits();
+ size_t digits = 0;
+ bigint bigmant;
+ parse_mantissa(bigmant, num, max_digits, digits);
+ // can't underflow, since digits is at most max_digits.
+ int32_t exponent = sci_exp + 1 - int32_t(digits);
+ if (exponent >= 0) {
+ return positive_digit_comp<T>(bigmant, exponent);
+ } else {
+ return negative_digit_comp<T>(bigmant, am, exponent);
}
}
-uint32_t number_of_digits_decimal_left_shift(decimal &h, uint32_t shift) {
- shift &= 63;
- const static uint16_t number_of_digits_decimal_left_shift_table[65] = {
- 0x0000, 0x0800, 0x0801, 0x0803, 0x1006, 0x1009, 0x100D, 0x1812, 0x1817,
- 0x181D, 0x2024, 0x202B, 0x2033, 0x203C, 0x2846, 0x2850, 0x285B, 0x3067,
- 0x3073, 0x3080, 0x388E, 0x389C, 0x38AB, 0x38BB, 0x40CC, 0x40DD, 0x40EF,
- 0x4902, 0x4915, 0x4929, 0x513E, 0x5153, 0x5169, 0x5180, 0x5998, 0x59B0,
- 0x59C9, 0x61E3, 0x61FD, 0x6218, 0x6A34, 0x6A50, 0x6A6D, 0x6A8B, 0x72AA,
- 0x72C9, 0x72E9, 0x7B0A, 0x7B2B, 0x7B4D, 0x8370, 0x8393, 0x83B7, 0x83DC,
- 0x8C02, 0x8C28, 0x8C4F, 0x9477, 0x949F, 0x94C8, 0x9CF2, 0x051C, 0x051C,
- 0x051C, 0x051C,
- };
- uint32_t x_a = number_of_digits_decimal_left_shift_table[shift];
- uint32_t x_b = number_of_digits_decimal_left_shift_table[shift + 1];
- uint32_t num_new_digits = x_a >> 11;
- uint32_t pow5_a = 0x7FF & x_a;
- uint32_t pow5_b = 0x7FF & x_b;
- const static uint8_t
- number_of_digits_decimal_left_shift_table_powers_of_5[0x051C] = {
- 5, 2, 5, 1, 2, 5, 6, 2, 5, 3, 1, 2, 5, 1, 5, 6, 2, 5, 7, 8, 1, 2, 5,
- 3, 9, 0, 6, 2, 5, 1, 9, 5, 3, 1, 2, 5, 9, 7, 6, 5, 6, 2, 5, 4, 8, 8,
- 2, 8, 1, 2, 5, 2, 4, 4, 1, 4, 0, 6, 2, 5, 1, 2, 2, 0, 7, 0, 3, 1, 2,
- 5, 6, 1, 0, 3, 5, 1, 5, 6, 2, 5, 3, 0, 5, 1, 7, 5, 7, 8, 1, 2, 5, 1,
- 5, 2, 5, 8, 7, 8, 9, 0, 6, 2, 5, 7, 6, 2, 9, 3, 9, 4, 5, 3, 1, 2, 5,
- 3, 8, 1, 4, 6, 9, 7, 2, 6, 5, 6, 2, 5, 1, 9, 0, 7, 3, 4, 8, 6, 3, 2,
- 8, 1, 2, 5, 9, 5, 3, 6, 7, 4, 3, 1, 6, 4, 0, 6, 2, 5, 4, 7, 6, 8, 3,
- 7, 1, 5, 8, 2, 0, 3, 1, 2, 5, 2, 3, 8, 4, 1, 8, 5, 7, 9, 1, 0, 1, 5,
- 6, 2, 5, 1, 1, 9, 2, 0, 9, 2, 8, 9, 5, 5, 0, 7, 8, 1, 2, 5, 5, 9, 6,
- 0, 4, 6, 4, 4, 7, 7, 5, 3, 9, 0, 6, 2, 5, 2, 9, 8, 0, 2, 3, 2, 2, 3,
- 8, 7, 6, 9, 5, 3, 1, 2, 5, 1, 4, 9, 0, 1, 1, 6, 1, 1, 9, 3, 8, 4, 7,
- 6, 5, 6, 2, 5, 7, 4, 5, 0, 5, 8, 0, 5, 9, 6, 9, 2, 3, 8, 2, 8, 1, 2,
- 5, 3, 7, 2, 5, 2, 9, 0, 2, 9, 8, 4, 6, 1, 9, 1, 4, 0, 6, 2, 5, 1, 8,
- 6, 2, 6, 4, 5, 1, 4, 9, 2, 3, 0, 9, 5, 7, 0, 3, 1, 2, 5, 9, 3, 1, 3,
- 2, 2, 5, 7, 4, 6, 1, 5, 4, 7, 8, 5, 1, 5, 6, 2, 5, 4, 6, 5, 6, 6, 1,
- 2, 8, 7, 3, 0, 7, 7, 3, 9, 2, 5, 7, 8, 1, 2, 5, 2, 3, 2, 8, 3, 0, 6,
- 4, 3, 6, 5, 3, 8, 6, 9, 6, 2, 8, 9, 0, 6, 2, 5, 1, 1, 6, 4, 1, 5, 3,
- 2, 1, 8, 2, 6, 9, 3, 4, 8, 1, 4, 4, 5, 3, 1, 2, 5, 5, 8, 2, 0, 7, 6,
- 6, 0, 9, 1, 3, 4, 6, 7, 4, 0, 7, 2, 2, 6, 5, 6, 2, 5, 2, 9, 1, 0, 3,
- 8, 3, 0, 4, 5, 6, 7, 3, 3, 7, 0, 3, 6, 1, 3, 2, 8, 1, 2, 5, 1, 4, 5,
- 5, 1, 9, 1, 5, 2, 2, 8, 3, 6, 6, 8, 5, 1, 8, 0, 6, 6, 4, 0, 6, 2, 5,
- 7, 2, 7, 5, 9, 5, 7, 6, 1, 4, 1, 8, 3, 4, 2, 5, 9, 0, 3, 3, 2, 0, 3,
- 1, 2, 5, 3, 6, 3, 7, 9, 7, 8, 8, 0, 7, 0, 9, 1, 7, 1, 2, 9, 5, 1, 6,
- 6, 0, 1, 5, 6, 2, 5, 1, 8, 1, 8, 9, 8, 9, 4, 0, 3, 5, 4, 5, 8, 5, 6,
- 4, 7, 5, 8, 3, 0, 0, 7, 8, 1, 2, 5, 9, 0, 9, 4, 9, 4, 7, 0, 1, 7, 7,
- 2, 9, 2, 8, 2, 3, 7, 9, 1, 5, 0, 3, 9, 0, 6, 2, 5, 4, 5, 4, 7, 4, 7,
- 3, 5, 0, 8, 8, 6, 4, 6, 4, 1, 1, 8, 9, 5, 7, 5, 1, 9, 5, 3, 1, 2, 5,
- 2, 2, 7, 3, 7, 3, 6, 7, 5, 4, 4, 3, 2, 3, 2, 0, 5, 9, 4, 7, 8, 7, 5,
- 9, 7, 6, 5, 6, 2, 5, 1, 1, 3, 6, 8, 6, 8, 3, 7, 7, 2, 1, 6, 1, 6, 0,
- 2, 9, 7, 3, 9, 3, 7, 9, 8, 8, 2, 8, 1, 2, 5, 5, 6, 8, 4, 3, 4, 1, 8,
- 8, 6, 0, 8, 0, 8, 0, 1, 4, 8, 6, 9, 6, 8, 9, 9, 4, 1, 4, 0, 6, 2, 5,
- 2, 8, 4, 2, 1, 7, 0, 9, 4, 3, 0, 4, 0, 4, 0, 0, 7, 4, 3, 4, 8, 4, 4,
- 9, 7, 0, 7, 0, 3, 1, 2, 5, 1, 4, 2, 1, 0, 8, 5, 4, 7, 1, 5, 2, 0, 2,
- 0, 0, 3, 7, 1, 7, 4, 2, 2, 4, 8, 5, 3, 5, 1, 5, 6, 2, 5, 7, 1, 0, 5,
- 4, 2, 7, 3, 5, 7, 6, 0, 1, 0, 0, 1, 8, 5, 8, 7, 1, 1, 2, 4, 2, 6, 7,
- 5, 7, 8, 1, 2, 5, 3, 5, 5, 2, 7, 1, 3, 6, 7, 8, 8, 0, 0, 5, 0, 0, 9,
- 2, 9, 3, 5, 5, 6, 2, 1, 3, 3, 7, 8, 9, 0, 6, 2, 5, 1, 7, 7, 6, 3, 5,
- 6, 8, 3, 9, 4, 0, 0, 2, 5, 0, 4, 6, 4, 6, 7, 7, 8, 1, 0, 6, 6, 8, 9,
- 4, 5, 3, 1, 2, 5, 8, 8, 8, 1, 7, 8, 4, 1, 9, 7, 0, 0, 1, 2, 5, 2, 3,
- 2, 3, 3, 8, 9, 0, 5, 3, 3, 4, 4, 7, 2, 6, 5, 6, 2, 5, 4, 4, 4, 0, 8,
- 9, 2, 0, 9, 8, 5, 0, 0, 6, 2, 6, 1, 6, 1, 6, 9, 4, 5, 2, 6, 6, 7, 2,
- 3, 6, 3, 2, 8, 1, 2, 5, 2, 2, 2, 0, 4, 4, 6, 0, 4, 9, 2, 5, 0, 3, 1,
- 3, 0, 8, 0, 8, 4, 7, 2, 6, 3, 3, 3, 6, 1, 8, 1, 6, 4, 0, 6, 2, 5, 1,
- 1, 1, 0, 2, 2, 3, 0, 2, 4, 6, 2, 5, 1, 5, 6, 5, 4, 0, 4, 2, 3, 6, 3,
- 1, 6, 6, 8, 0, 9, 0, 8, 2, 0, 3, 1, 2, 5, 5, 5, 5, 1, 1, 1, 5, 1, 2,
- 3, 1, 2, 5, 7, 8, 2, 7, 0, 2, 1, 1, 8, 1, 5, 8, 3, 4, 0, 4, 5, 4, 1,
- 0, 1, 5, 6, 2, 5, 2, 7, 7, 5, 5, 5, 7, 5, 6, 1, 5, 6, 2, 8, 9, 1, 3,
- 5, 1, 0, 5, 9, 0, 7, 9, 1, 7, 0, 2, 2, 7, 0, 5, 0, 7, 8, 1, 2, 5, 1,
- 3, 8, 7, 7, 7, 8, 7, 8, 0, 7, 8, 1, 4, 4, 5, 6, 7, 5, 5, 2, 9, 5, 3,
- 9, 5, 8, 5, 1, 1, 3, 5, 2, 5, 3, 9, 0, 6, 2, 5, 6, 9, 3, 8, 8, 9, 3,
- 9, 0, 3, 9, 0, 7, 2, 2, 8, 3, 7, 7, 6, 4, 7, 6, 9, 7, 9, 2, 5, 5, 6,
- 7, 6, 2, 6, 9, 5, 3, 1, 2, 5, 3, 4, 6, 9, 4, 4, 6, 9, 5, 1, 9, 5, 3,
- 6, 1, 4, 1, 8, 8, 8, 2, 3, 8, 4, 8, 9, 6, 2, 7, 8, 3, 8, 1, 3, 4, 7,
- 6, 5, 6, 2, 5, 1, 7, 3, 4, 7, 2, 3, 4, 7, 5, 9, 7, 6, 8, 0, 7, 0, 9,
- 4, 4, 1, 1, 9, 2, 4, 4, 8, 1, 3, 9, 1, 9, 0, 6, 7, 3, 8, 2, 8, 1, 2,
- 5, 8, 6, 7, 3, 6, 1, 7, 3, 7, 9, 8, 8, 4, 0, 3, 5, 4, 7, 2, 0, 5, 9,
- 6, 2, 2, 4, 0, 6, 9, 5, 9, 5, 3, 3, 6, 9, 1, 4, 0, 6, 2, 5,
- };
- const uint8_t *pow5 =
- &number_of_digits_decimal_left_shift_table_powers_of_5[pow5_a];
- uint32_t i = 0;
- uint32_t n = pow5_b - pow5_a;
- for (; i < n; i++) {
- if (i >= h.num_digits) {
- return num_new_digits - 1;
- } else if (h.digits[i] == pow5[i]) {
- continue;
- } else if (h.digits[i] < pow5[i]) {
- return num_new_digits - 1;
- } else {
- return num_new_digits;
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+#define SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+
+
+#include <cmath>
+#include <cstring>
+#include <limits>
+#include <system_error>
+
+namespace simdjson_fast_float {
+
+namespace detail {
+/**
+ * Special case +inf, -inf, nan, infinity, -infinity.
+ * The case comparisons could be made much faster given that we know that the
+ * strings a null-free and fixed.
+ **/
+template <typename T, typename UC>
+from_chars_result_t<UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 parse_infnan(UC const *first, UC const *last,
+ T &value, chars_format fmt) noexcept {
+ from_chars_result_t<UC> answer{};
+ answer.ptr = first;
+ answer.ec = std::errc(); // be optimistic
+ // assume first < last, so dereference without checks;
+ bool const minusSign = (*first == UC('-'));
+ // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+ if ((*first == UC('-')) ||
+ (uint64_t(fmt & chars_format::allow_leading_plus) &&
+ (*first == UC('+')))) {
+ ++first;
+ }
+ if (last - first >= 3) {
+ if (simdjson_fastfloat_strncasecmp3(first, str_const_nan<UC>())) {
+ answer.ptr = (first += 3);
+ value = minusSign ? -std::numeric_limits<T>::quiet_NaN()
+ : std::numeric_limits<T>::quiet_NaN();
+ // Check for possible nan(n-char-seq-opt), C++17 20.19.3.7,
+ // C11 7.20.1.3.3. At least MSVC produces nan(ind) and nan(snan).
+ if (first != last && *first == UC('(')) {
+ for (UC const *ptr = first + 1; ptr != last; ++ptr) {
+ if (*ptr == UC(')')) {
+ answer.ptr = ptr + 1; // valid nan(n-char-seq-opt)
+ break;
+ } else if (!((UC('a') <= *ptr && *ptr <= UC('z')) ||
+ (UC('A') <= *ptr && *ptr <= UC('Z')) ||
+ (UC('0') <= *ptr && *ptr <= UC('9')) || *ptr == UC('_')))
+ break; // forbidden char, not nan(n-char-seq-opt)
+ }
+ }
+ return answer;
+ }
+ if (simdjson_fastfloat_strncasecmp3(first, str_const_inf<UC>())) {
+ if ((last - first >= 8) &&
+ simdjson_fastfloat_strncasecmp5(first + 3, str_const_inf<UC>() + 3)) {
+ answer.ptr = first + 8;
+ } else {
+ answer.ptr = first + 3;
+ }
+ value = minusSign ? -std::numeric_limits<T>::infinity()
+ : std::numeric_limits<T>::infinity();
+ return answer;
}
}
- return num_new_digits;
+ answer.ec = std::errc::invalid_argument;
+ return answer;
+}
+
+/**
+ * Returns true if the floating-pointing rounding mode is to 'nearest'.
+ * It is the default on most system. This function is meant to be inexpensive.
+ * Credit : @mwalcott3
+ */
+simdjson_fastfloat_really_inline bool rounds_to_nearest() noexcept {
+ // https://lemire.me/blog/2020/06/26/gcc-not-nearest/
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return false;
+#endif
+ // See
+ // A fast function to check your floating-point rounding mode
+ // https://lemire.me/blog/2022/11/16/a-fast-function-to-check-your-floating-point-rounding-mode/
+ //
+ // This function is meant to be equivalent to :
+ // prior: #include <cfenv>
+ // return fegetround() == FE_TONEAREST;
+ // However, it is expected to be much faster than the fegetround()
+ // function call.
+ //
+ // The volatile keyword prevents the compiler from computing the function
+ // at compile-time.
+ // There might be other ways to prevent compile-time optimizations (e.g.,
+ // asm). The value does not need to be std::numeric_limits<float>::min(), any
+ // small value so that 1 + x should round to 1 would do (after accounting for
+ // excess precision, as in 387 instructions).
+ static float volatile fmin = (std::numeric_limits<float>::min)();
+ float fmini = fmin; // we copy it so that it gets loaded at most once.
+//
+// Explanation:
+// Only when fegetround() == FE_TONEAREST do we have that
+// fmin + 1.0f == 1.0f - fmin.
+//
+// FE_UPWARD:
+// fmin + 1.0f > 1
+// 1.0f - fmin == 1
+//
+// FE_DOWNWARD or FE_TOWARDZERO:
+// fmin + 1.0f == 1
+// 1.0f - fmin < 1
+//
+// Note: This may fail to be accurate if fast-math has been
+// enabled, as rounding conventions may not apply.
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+// todo: is there a VS warning?
+// see
+// https://stackoverflow.com/questions/46079446/is-there-a-warning-for-floating-point-equality-checking-in-visual-studio-2013
+#elif defined(__clang__)
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Wfloat-equal"
+#elif defined(__GNUC__)
+#pragma GCC diagnostic push
+#pragma GCC diagnostic ignored "-Wfloat-equal"
+#endif
+ return (fmini + 1.0f == 1.0f - fmini);
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#elif defined(__clang__)
+#pragma clang diagnostic pop
+#elif defined(__GNUC__)
+#pragma GCC diagnostic pop
+#endif
}
-} // end of anonymous namespace
+} // namespace detail
-uint64_t round(decimal &h) {
- if ((h.num_digits == 0) || (h.decimal_point < 0)) {
- return 0;
- } else if (h.decimal_point > 18) {
- return UINT64_MAX;
- }
- // at this point, we know that h.decimal_point >= 0
- uint32_t dp = uint32_t(h.decimal_point);
- uint64_t n = 0;
- for (uint32_t i = 0; i < dp; i++) {
- n = (10 * n) + ((i < h.num_digits) ? h.digits[i] : 0);
+template <typename T> struct from_chars_caller {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_advanced(first, last, value, options);
}
- bool round_up = false;
- if (dp < h.num_digits) {
- round_up = h.digits[dp] >= 5; // normally, we round up
- // but we may need to round to even!
- if ((h.digits[dp] == 5) && (dp + 1 == h.num_digits)) {
- round_up = h.truncated || ((dp > 0) && (1 & h.digits[dp - 1]));
- }
+};
+
+#ifdef __STDCPP_FLOAT32_T__
+template <> struct from_chars_caller<std::float32_t> {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, std::float32_t &value,
+ parse_options_t<UC> options) noexcept {
+ // if std::float32_t is defined, and we are in C++23 mode; macro set for
+ // float32; set value to float due to equivalence between float and
+ // float32_t
+ float val = 0.0f;
+ auto ret = from_chars_advanced(first, last, val, options);
+ value = val;
+ return ret;
}
- if (round_up) {
- n++;
+};
+#endif
+
+#ifdef __STDCPP_FLOAT64_T__
+template <> struct from_chars_caller<std::float64_t> {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, std::float64_t &value,
+ parse_options_t<UC> options) noexcept {
+ // if std::float64_t is defined, and we are in C++23 mode; macro set for
+ // float64; set value as double due to equivalence between double and
+ // float64_t
+ double val = 0.0;
+ auto ret = from_chars_advanced(first, last, val, options);
+ value = val;
+ return ret;
}
- return n;
+};
+#endif
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+ chars_format fmt /*= chars_format::general*/) noexcept {
+ return from_chars_caller<T>::call(first, last, value,
+ parse_options_t<UC>(fmt));
}
-// computes h * 2^-shift
-void decimal_left_shift(decimal &h, uint32_t shift) {
- if (h.num_digits == 0) {
- return;
- }
- uint32_t num_new_digits = number_of_digits_decimal_left_shift(h, shift);
- int32_t read_index = int32_t(h.num_digits - 1);
- uint32_t write_index = h.num_digits - 1 + num_new_digits;
- uint64_t n = 0;
-
- while (read_index >= 0) {
- n += uint64_t(h.digits[read_index]) << shift;
- uint64_t quotient = n / 10;
- uint64_t remainder = n - (10 * quotient);
- if (write_index < max_digits) {
- h.digits[write_index] = uint8_t(remainder);
- } else if (remainder > 0) {
- h.truncated = true;
- }
- n = quotient;
- write_index--;
- read_index--;
- }
- while (n > 0) {
- uint64_t quotient = n / 10;
- uint64_t remainder = n - (10 * quotient);
- if (write_index < max_digits) {
- h.digits[write_index] = uint8_t(remainder);
- } else if (remainder > 0) {
- h.truncated = true;
- }
- n = quotient;
- write_index--;
- }
- h.num_digits += num_new_digits;
- if (h.num_digits > max_digits) {
- h.num_digits = max_digits;
- }
- h.decimal_point += int32_t(num_new_digits);
- trim(h);
-}
-
-// computes h * 2^shift
-void decimal_right_shift(decimal &h, uint32_t shift) {
- uint32_t read_index = 0;
- uint32_t write_index = 0;
-
- uint64_t n = 0;
-
- while ((n >> shift) == 0) {
- if (read_index < h.num_digits) {
- n = (10 * n) + h.digits[read_index++];
- } else if (n == 0) {
- return;
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+clinger_fast_path_impl(uint64_t mantissa, int64_t exponent, bool is_negative,
+ T &value) noexcept {
+ // The implementation of the Clinger's fast path is convoluted because
+ // we want round-to-nearest in all cases, irrespective of the rounding mode
+ // selected on the thread.
+ // We proceed optimistically, assuming that detail::rounds_to_nearest()
+ // returns true.
+ if (binary_format<T>::min_exponent_fast_path() <= exponent &&
+ exponent <= binary_format<T>::max_exponent_fast_path() &&
+ mantissa <= binary_format<T>::max_mantissa_fast_path()) {
+ // The mantissa bound above is a necessary condition for BOTH branches
+ // below: the rounding-mode-dependent branch checks the tighter
+ // max_mantissa_fast_path(exponent) <= max_mantissa_fast_path(). Testing
+ // it before detail::rounds_to_nearest() spares long-mantissa inputs
+ // (which can never take the fast path) the volatile-float probe.
+ //
+ // Unfortunately, the conventional Clinger's fast path is only possible
+ // when the system rounds to the nearest float.
+ //
+ // We expect the next branch to almost always be selected.
+ // We could check it first (before the previous branch), but
+ // there might be performance advantages at having the check
+ // be last.
+ if (!cpp20_and_in_constexpr() && detail::rounds_to_nearest()) {
+ // We have that fegetround() == FE_TONEAREST.
+ // Next is Clinger's fast path.
+ value = T(mantissa);
+ if (exponent < 0) {
+ value = value / binary_format<T>::exact_power_of_ten(-exponent);
+ } else {
+ value = value * binary_format<T>::exact_power_of_ten(exponent);
+ }
+ if (is_negative) {
+ value = -value;
+ }
+ return true;
} else {
- while ((n >> shift) == 0) {
- n = 10 * n;
- read_index++;
+ // We do not have that fegetround() == FE_TONEAREST.
+ // Next is a modified Clinger's fast path, inspired by Jakub Jelinek's
+ // proposal
+ if (exponent >= 0 &&
+ mantissa <= binary_format<T>::max_mantissa_fast_path(exponent)) {
+#if defined(__clang__) || defined(SIMDJSON_FASTFLOAT_32BIT)
+ // Clang may map 0 to -0.0 when fegetround() == FE_DOWNWARD
+ if (mantissa == 0) {
+ value = is_negative ? T(-0.) : T(0.);
+ return true;
+ }
+#endif
+ value = T(mantissa) * binary_format<T>::exact_power_of_ten(exponent);
+ if (is_negative) {
+ value = -value;
+ }
+ return true;
}
- break;
}
}
- h.decimal_point -= int32_t(read_index - 1);
- if (h.decimal_point < -decimal_point_range) { // it is zero
- h.num_digits = 0;
- h.decimal_point = 0;
- h.negative = false;
- h.truncated = false;
- return;
+ return false;
+}
+
+/**
+ * This function overload takes parsed_number_string_t structure that is created
+ * and populated either by from_chars_advanced function taking chars range and
+ * parsing options or other parsing custom function implemented by user.
+ */
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(parsed_number_string_t<UC> &pns, T &value) noexcept {
+ static_assert(is_supported_float_type<T>::value,
+ "only some floating-point types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ from_chars_result_t<UC> answer;
+
+ answer.ec = std::errc(); // be optimistic
+ answer.ptr = pns.lastmatch;
+
+ if (!pns.too_many_digits &&
+ clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value))
+ return answer;
+
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+ if (pns.too_many_digits && am.power2 >= 0) {
+ if (am != compute_float<binary_format<T>>(pns.exponent, pns.mantissa + 1)) {
+ am = compute_error<binary_format<T>>(pns.exponent, pns.mantissa);
+ }
}
- uint64_t mask = (uint64_t(1) << shift) - 1;
- while (read_index < h.num_digits) {
- uint8_t new_digit = uint8_t(n >> shift);
- n = (10 * (n & mask)) + h.digits[read_index++];
- h.digits[write_index++] = new_digit;
+ // If we called compute_float<binary_format<T>>(pns.exponent, pns.mantissa)
+ // and we have an invalid power (am.power2 < 0), then we need to go the long
+ // way around again. This is very uncommon.
+ if (am.power2 < 0) {
+ am = digit_comp<T>(pns, am);
}
- while (n > 0) {
- uint8_t new_digit = uint8_t(n >> shift);
- n = 10 * (n & mask);
- if (write_index < max_digits) {
- h.digits[write_index++] = new_digit;
- } else if (new_digit > 0) {
- h.truncated = true;
- }
+ to_float(pns.negative, am, value);
+ // Test for over/underflow.
+ if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+ am.power2 == binary_format<T>::infinite_power()) {
+ answer.ec = std::errc::result_out_of_range;
}
- h.num_digits = write_index;
- trim(h);
+ return answer;
}
-template <typename binary> adjusted_mantissa compute_float(decimal &d) {
- adjusted_mantissa answer;
- if (d.num_digits == 0) {
- // should be zero
- answer.power2 = 0;
- answer.mantissa = 0;
- return answer;
- }
- // At this point, going further, we can assume that d.num_digits > 0.
- // We want to guard against excessive decimal point values because
- // they can result in long running times. Indeed, we do
- // shifts by at most 60 bits. We have that log(10**400)/log(2**60) ~= 22
- // which is fine, but log(10**299995)/log(2**60) ~= 16609 which is not
- // fine (runs for a long time).
- //
- if(d.decimal_point < -324) {
- // We have something smaller than 1e-324 which is always zero
- // in binary64 and binary32.
- // It should be zero.
- answer.power2 = 0;
- answer.mantissa = 0;
- return answer;
- } else if(d.decimal_point >= 310) {
- // We have something at least as large as 0.1e310 which is
- // always infinite.
- answer.power2 = binary::infinite_power();
- answer.mantissa = 0;
+// Slow path: re-parse materializing the integer/fraction spans the hot no-span
+// parse skipped, then run the full algorithm. The two callers reach it only
+// through a simdjson_fastfloat_unlikely branch, so the optimizer keeps this re-parse off
+// the hot path on its own (no function-level noinline needed).
+// from_chars_advanced already handles both the too_many_digits disambiguation
+// and the am.power2<0 digit_comp recompute, so both slow branches collapse to
+// one helper call.
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_number_slow_path(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options, bool bjf) noexcept {
+ parsed_number_string_t<UC> pns =
+ bjf ? parse_number_string<true, UC>(first, last, options, true)
+ : parse_number_string<false, UC>(first, last, options, true);
+ return from_chars_advanced(pns, value);
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_float_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+
+ static_assert(is_supported_float_type<T>::value,
+ "only some floating-point types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+
+ from_chars_result_t<UC> answer;
+ if (uint64_t(fmt & chars_format::skip_white_space)) {
+ while ((first != last) && simdjson_fast_float::is_space(*first)) {
+ first++;
+ }
+ }
+ if (first == last) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
return answer;
}
-
- static const uint32_t max_shift = 60;
- static const uint32_t num_powers = 19;
- static const uint8_t powers[19] = {
- 0, 3, 6, 9, 13, 16, 19, 23, 26, 29, //
- 33, 36, 39, 43, 46, 49, 53, 56, 59, //
- };
- int32_t exp2 = 0;
- while (d.decimal_point > 0) {
- uint32_t n = uint32_t(d.decimal_point);
- uint32_t shift = (n < num_powers) ? powers[n] : max_shift;
- decimal_right_shift(d, shift);
- if (d.decimal_point < -decimal_point_range) {
- // should be zero
- answer.power2 = 0;
- answer.mantissa = 0;
+ bool const bjf = uint64_t(fmt & detail::basic_json_fmt) != 0;
+
+ // Fast path: parse WITHOUT materializing the integer/fraction spans (read
+ // only by the rare slow paths). Skipping their stores keeps the fat
+ // parsed_number_string_t off the hot path. store_spans is a runtime argument,
+ // so this reuses the single parse_number_string instantiation.
+ parsed_number_string_t<UC> pns =
+ bjf ? parse_number_string<true, UC>(first, last, options, false)
+ : parse_number_string<false, UC>(first, last, options, false);
+ if (!pns.valid) {
+ if (uint64_t(fmt & chars_format::no_infnan)) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
return answer;
- }
- exp2 += int32_t(shift);
- }
- // We shift left toward [1/2 ... 1].
- while (d.decimal_point <= 0) {
- uint32_t shift;
- if (d.decimal_point == 0) {
- if (d.digits[0] >= 5) {
- break;
- }
- shift = (d.digits[0] < 2) ? 2 : 1;
} else {
- uint32_t n = uint32_t(-d.decimal_point);
- shift = (n < num_powers) ? powers[n] : max_shift;
+ return detail::parse_infnan(first, last, value, fmt);
}
- decimal_left_shift(d, shift);
- if (d.decimal_point > decimal_point_range) {
- // we want to get infinity:
- answer.power2 = 0xFF;
- answer.mantissa = 0;
- return answer;
- }
- exp2 -= int32_t(shift);
}
- // We are now in the range [1/2 ... 1] but the binary format uses [1 ... 2].
- exp2--;
- constexpr int32_t minimum_exponent = binary::minimum_exponent();
- while ((minimum_exponent + 1) > exp2) {
- uint32_t n = uint32_t((minimum_exponent + 1) - exp2);
- if (n > max_shift) {
- n = max_shift;
- }
- decimal_right_shift(d, n);
- exp2 += int32_t(n);
+
+ // Slow path A (rare): > 19 significant digits. The no-span parse left the
+ // mantissa un-truncated and skipped the span-based recompute; the cold helper
+ // re-parses with spans and runs the full algorithm.
+ //
+// We have to disable -Wc++20-extensions for the [[unlikely]] attribute
+// See comment for @jwakely at
+// https://github.com/fastfloat/simdjson_fast_float/pull/387#discussion_r3366943539
+// This is unfortunate.
+#ifdef __clang__
+#pragma clang diagnostic push
+#if (!defined(__APPLE_CC__) && __clang_major__ >= 10) || (__clang_major__ >= 13)
+#pragma clang diagnostic ignored "-Wc++20-extensions"
+#endif
+#endif
+ if simdjson_fastfloat_unlikely (pns.too_many_digits) {
+ return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
}
- if ((exp2 - minimum_exponent) >= binary::infinite_power()) {
- answer.power2 = binary::infinite_power();
- answer.mantissa = 0;
+ answer.ec = std::errc(); // be optimistic
+ answer.ptr = pns.lastmatch;
+
+ if (clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value)) {
return answer;
}
- const int mantissa_size_in_bits = binary::mantissa_explicit_bits() + 1;
- decimal_left_shift(d, mantissa_size_in_bits);
-
- uint64_t mantissa = round(d);
- // It is possible that we have an overflow, in which case we need
- // to shift back.
- if (mantissa >= (uint64_t(1) << mantissa_size_in_bits)) {
- decimal_right_shift(d, 1);
- exp2 += 1;
- mantissa = round(d);
- if ((exp2 - minimum_exponent) >= binary::infinite_power()) {
- answer.power2 = binary::infinite_power();
- answer.mantissa = 0;
- return answer;
- }
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+ // Slow path B (rare): Eisel-Lemire could not resolve; digit_comp needs the
+ // integer/fraction spans. Route to the cold helper (clinger there is a
+ // dead-effect since it already failed here; the cold re-parse + digit_comp
+ // via from_chars_advanced reproduces this branch).
+ if simdjson_fastfloat_unlikely (am.power2 < 0) {
+ return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
}
- answer.power2 = exp2 - binary::minimum_exponent();
- if (mantissa < (uint64_t(1) << binary::mantissa_explicit_bits())) {
- answer.power2--;
+#ifdef __clang__
+#pragma clang diagnostic pop
+#endif
+ to_float(pns.negative, am, value);
+ // Test for over/underflow.
+ if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+ am.power2 == binary_format<T>::infinite_power()) {
+ answer.ec = std::errc::result_out_of_range;
}
- answer.mantissa =
- mantissa & ((uint64_t(1) << binary::mantissa_explicit_bits()) - 1);
return answer;
}
-template <typename binary>
-adjusted_mantissa parse_long_mantissa(const char *first) {
- decimal d = parse_decimal(first);
- return compute_float<binary>(d);
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base) noexcept {
+
+ static_assert(is_supported_integer_type<T>::value,
+ "only integer types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ parse_options_t<UC> options;
+ options.base = base;
+ return from_chars_advanced(first, last, value, options);
}
-template <typename binary>
-adjusted_mantissa parse_long_mantissa(const char *first, const char *end) {
- decimal d = parse_decimal(first, end);
- return compute_float<binary>(d);
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+ T value;
+ if (clinger_fast_path_impl(mantissa, decimal_exponent, false, value))
+ return value;
+
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(decimal_exponent, mantissa);
+ to_float(false, am, value);
+ return value;
}
-double from_chars(const char *first) noexcept {
- bool negative = first[0] == '-';
- if (negative) {
- first++;
- }
- adjusted_mantissa am = parse_long_mantissa<binary_format<double>>(first);
- uint64_t word = am.mantissa;
- word |= uint64_t(am.power2)
- << binary_format<double>::mantissa_explicit_bits();
- word = negative ? word | (uint64_t(1) << binary_format<double>::sign_index())
- : word;
- double value;
- std::memcpy(&value, &word, sizeof(double));
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+ const bool is_negative = mantissa < 0;
+ const uint64_t m = static_cast<uint64_t>(is_negative ? -mantissa : mantissa);
+
+ T value;
+ if (clinger_fast_path_impl(m, decimal_exponent, is_negative, value))
+ return value;
+
+ adjusted_mantissa am = compute_float<binary_format<T>>(decimal_exponent, m);
+ to_float(is_negative, am, value);
return value;
}
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
-double from_chars(const char *first, const char *end) noexcept {
- bool negative = first[0] == '-';
- if (negative) {
- first++;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+// the following overloads are here to avoid surprising ambiguity for int,
+// unsigned, etc.
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value &&
+ std::is_integral<Int>::value &&
+ !std::is_signed<Int>::value,
+ T>::type
+ integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<T>(static_cast<uint64_t>(mantissa),
+ decimal_exponent);
+}
+
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value &&
+ std::is_integral<Int>::value &&
+ std::is_signed<Int>::value,
+ T>::type
+ integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<T>(static_cast<int64_t>(mantissa),
+ decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+ std::is_integral<Int>::value && !std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10(static_cast<uint64_t>(mantissa), decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+ std::is_integral<Int>::value && std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10(static_cast<int64_t>(mantissa), decimal_exponent);
+}
+
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_int_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+
+ static_assert(is_supported_integer_type<T>::value,
+ "only integer types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ int const base = options.base;
+
+ from_chars_result_t<UC> answer;
+ if (uint64_t(fmt & chars_format::skip_white_space)) {
+ while ((first != last) && simdjson_fast_float::is_space(*first)) {
+ first++;
+ }
+ }
+ if (first == last || base < 2 || base > 36) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+
+ return parse_int_string(first, last, value, options);
+}
+
+template <size_t TypeIx> struct from_chars_advanced_caller {
+ static_assert(TypeIx > 0, "unsupported type");
+};
+
+template <> struct from_chars_advanced_caller<1> {
+ template <typename T, typename UC>
+ simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_float_advanced(first, last, value, options);
+ }
+};
+
+template <> struct from_chars_advanced_caller<2> {
+ template <typename T, typename UC>
+ simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_int_advanced(first, last, value, options);
+ }
+};
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_advanced_caller<
+ size_t(is_supported_float_type<T>::value) +
+ 2 * size_t(is_supported_integer_type<T>::value)>::call(first, last, value,
+ options);
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+/* end file simdjson/internal/fast_float.h */
+
+#include <cstdint>
+#include <cstring>
+#include <limits>
+
+namespace simdjson {
+namespace internal {
+
+/**
+ * These functions handle floating-point parsing when the fast path in
+ * numberparsing.h gives up: more than 19 digits in the decimal mantissa, an
+ * exponent outside the range the power-of-five table covers, or one of the rare
+ * inputs where the truncated Eisel-Lemire product is not accurate enough to
+ * round. That should only be seen in adversarial scenarios; we do not expect
+ * production systems to even produce such floating-point numbers.
+ *
+ * The work is handed to fast_float (vendored in
+ * include/simdjson/internal/fast_float.h), which settles the rounding by
+ * comparing a bigint against a scaled power of five. It is correctly rounded,
+ * and quick enough that an adversarial document is no longer worth worrying
+ * about.
+ **/
+
+namespace {
+
+// fast_float wants the end of the number, and the callers only promise that a
+// number is followed by a character which cannot be part of one -- the input
+// has already been validated against the JSON grammar, and the buffer is padded,
+// so such a character is always there to be found. Locating it costs a pass over
+// digits we are about to parse anyway, and in exchange fast_float can bound its
+// inner loops instead of re-checking a far-away end pointer.
+const char *find_end_of_number(const char *first) noexcept {
+ const char *p = first;
+ while ((*p >= '0' && *p <= '9') || *p == '-' || *p == '+' || *p == '.' ||
+ *p == 'e' || *p == 'E') {
+ p++;
}
- adjusted_mantissa am = parse_long_mantissa<binary_format<double>>(first, end);
- uint64_t word = am.mantissa;
- word |= uint64_t(am.power2)
- << binary_format<double>::mantissa_explicit_bits();
- word = negative ? word | (uint64_t(1) << binary_format<double>::sign_index())
- : word;
- double value;
- std::memcpy(&value, &word, sizeof(double));
+ return p;
+}
+
+// The input is JSON, so parse it under the JSON grammar: no hexadecimal, no
+// leading plus, and no infinity or NaN spellings. Those are handled (or
+// rejected) before we ever get here.
+constexpr simdjson_fast_float::parse_options json_options{
+ simdjson_fast_float::chars_format::json};
+
+// fast_float reports result_out_of_range for a value at either edge of the
+// format, writing +/-0 when it underflows and +/-infinity when it overflows.
+// Both are exactly what the callers of these functions expect to receive: they
+// accept a zero and treat an infinity as an error. A malformed number cannot
+// happen on validated input, but if it somehow did, returning zero matches what
+// the previous implementation did with digits it could not use.
+template <typename T> T parse_with_fast_float(const char *first, const char *end) noexcept {
+ T value{};
+ auto answer =
+ simdjson_fast_float::from_chars_advanced(first, end, value, json_options);
+ if (answer.ec == std::errc::invalid_argument) { return T(0); }
return value;
}
+} // namespace
+
+double from_chars(const char *first) noexcept {
+ return parse_with_fast_float<double>(first, find_end_of_number(first));
+}
+
+double from_chars(const char *first, const char *end) noexcept {
+ return parse_with_fast_float<double>(first, end);
+}
+
+float from_chars_float(const char *first) noexcept {
+ return parse_with_fast_float<float>(first, find_end_of_number(first));
+}
+
} // internal
} // simdjson
@@ -5109,7 +10245,8 @@ namespace internal {
{ SCALAR_DOCUMENT_AS_VALUE, "SCALAR_DOCUMENT_AS_VALUE: A JSON document made of a scalar (number, Boolean, null or string) is treated as a value. Use get_bool(), get_double(), etc. on the document instead. "},
{ OUT_OF_BOUNDS, "OUT_OF_BOUNDS: Attempt to access location outside of document."},
{ TRAILING_CONTENT, "TRAILING_CONTENT: Unexpected trailing content in the JSON input."},
- { OUT_OF_CAPACITY, "OUT_OF_CAPACITY: The capacity was exceeded, we cannot allocate enough memory."}
+ { OUT_OF_CAPACITY, "OUT_OF_CAPACITY: The capacity was exceeded, we cannot allocate enough memory."},
+ { UNKNOWN_FIELD, "UNKNOWN_FIELD: The JSON object has a field that does not map to any member of the target type (deny_unknown_fields)."}
}; // error_messages[]
} // namespace internal
@@ -7115,7 +12252,12 @@ class document;
* 3) The stream_final mode allows us to truncate final
* unterminated strings. It is useful in conjunction with streaming_partial.
*/
-enum class stage1_mode { regular, streaming_partial, streaming_final};
+enum class stage1_mode {
+ regular,
+ streaming_partial, streaming_final,
+ json_sequence_partial, json_sequence_final,
+ comma_delimited_partial, comma_delimited_final
+};
/**
* Returns true if mode == streaming_partial or mode == streaming_final
@@ -7127,7 +12269,6 @@ inline bool is_streaming(stage1_mode mode) {
// return (mode == stage1_mode::streaming_partial || mode == stage1_mode::streaming_final);
}
-
namespace internal {
@@ -7312,6 +12453,16 @@ public:
/** Whether to store big integers as strings instead of returning BIGINT_ERROR */
bool _number_as_string{false};
+ /**
+ * Whether the input buffer passed to parse() is *not* padded to len +
+ * SIMDJSON_PADDING bytes. When true, stage 2 string parsing avoids reading
+ * past buf+len (it finishes the final, near-the-end bytes from a small padded
+ * scratch buffer). This is set only by the no-padding DOM parse entry points
+ * (dom::parser::parse_unpadded); the default padded fast path leaves it false
+ * and is unaffected.
+ */
+ bool _unpadded{false};
+
protected:
// Declaring these so that subclasses can use them to implement their constructors.
@@ -7655,6 +12806,8 @@ enum instruction_set {
LASX = 0x40000,
//RVV = 0x80000,
RVV_VLS = 0x100000,
+ SVE = 0x200000,
+ SVE2 = 0x400000,
};
} // namespace internal
@@ -7966,12 +13119,24 @@ POSSIBILITY OF SUCH DAMAGE.
#include <cstdlib>
#if defined(_MSC_VER)
#include <intrin.h>
-#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
+#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
#include <cpuid.h>
#endif
#if defined(__loongarch__) && defined(__linux__)
#include <sys/auxv.h>
#endif
+#if defined(__aarch64__) && defined(__linux__)
+ #include <sys/auxv.h>
+#endif
+#if (defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)) && defined(_WIN32) && !defined(_WINDOWS_)
+// We avoid including <windows.h> (macro pollution); this matches the
+// declaration in the Windows SDK (BOOL WINAPI IsProcessorFeaturePresent(DWORD)).
+extern "C" __declspec(dllimport) int __stdcall IsProcessorFeaturePresent(unsigned long ProcessorFeature);
+#endif
+
+#ifdef __FILC__
+#include <stdfil.h>
+#endif
namespace simdjson {
namespace internal {
@@ -7984,8 +13149,60 @@ static inline uint32_t detect_supported_architectures() {
#elif defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)
+#if defined(__linux__)
+// The kernel advertises SVE in AT_HWCAP and SVE2 in AT_HWCAP2. Older
+// headers may not define these constants, so we provide the kernel's values
+// (we deliberately do not include <asm/hwcap.h>, which is not available on
+// all toolchains, e.g., musl without linux-headers).
+#ifndef AT_HWCAP2
+#define AT_HWCAP2 26
+#endif
+#ifndef HWCAP_SVE
+#define HWCAP_SVE (1 << 22)
+#endif
+#ifndef HWCAP2_SVE2
+#define HWCAP2_SVE2 (1 << 1)
+#endif
+#endif // __linux__
+
+#if defined(_WIN32)
+// Only recent Windows SDKs define these processor features.
+#ifndef PF_ARM_SVE_INSTRUCTIONS_AVAILABLE
+#define PF_ARM_SVE_INSTRUCTIONS_AVAILABLE 46
+#endif
+#ifndef PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE
+#define PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE 47
+#endif
+#endif // _WIN32
+
static inline uint32_t detect_supported_architectures() {
- return instruction_set::NEON;
+ // NEON is mandatory on AArch64.
+ uint32_t host_isa = instruction_set::NEON;
+#if defined(__linux__)
+ unsigned long hwcap = getauxval(AT_HWCAP);
+ unsigned long hwcap2 = getauxval(AT_HWCAP2);
+ if (hwcap & HWCAP_SVE) {
+ host_isa |= instruction_set::SVE;
+ // We only claim SVE2 when SVE is also present. Before Linux 6.14, the
+ // kernel set HWCAP2_SVE2 on processors implementing SME(2) but not SVE,
+ // because SVE2 instructions are available in streaming mode. Our SVE2
+ // code runs in non-streaming mode and needs actual SVE.
+ if (hwcap2 & HWCAP2_SVE2) {
+ host_isa |= instruction_set::SVE2;
+ }
+ }
+#elif defined(_WIN32)
+ if (IsProcessorFeaturePresent(PF_ARM_SVE_INSTRUCTIONS_AVAILABLE)) {
+ host_isa |= instruction_set::SVE;
+ // As on Linux, require SVE before claiming SVE2.
+ if (IsProcessorFeaturePresent(PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE)) {
+ host_isa |= instruction_set::SVE2;
+ }
+ }
+#endif
+ // On other systems (e.g., macOS, where Apple Silicon has no SVE), we only
+ // report NEON.
+ return host_isa;
}
#elif defined(__x86_64__) || defined(_M_AMD64) // x64
@@ -8023,7 +13240,7 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
*ebx = cpu_info[1];
*ecx = cpu_info[2];
*edx = cpu_info[3];
-#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
+#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
uint32_t level = *eax;
__get_cpuid(level, eax, ebx, ecx, edx);
#else
@@ -8040,6 +13257,8 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
static inline uint64_t xgetbv() {
#if defined(_MSC_VER)
return _xgetbv(0);
+#elif defined(__FILC__)
+ return zxgetbv();
#else
uint32_t xcr0_lo, xcr0_hi;
asm volatile("xgetbv\n\t" : "=a" (xcr0_lo), "=d" (xcr0_hi) : "c" (0));
@@ -8601,7 +13820,7 @@ public:
simdjson_inline implementation() : simdjson::implementation(
"rvv_vls",
"RISC-V V extension",
- 0
+ internal::instruction_set::RVV_VLS
) {}
simdjson_warn_unused error_code create_dom_parser_implementation(
size_t capacity,
@@ -8976,7 +14195,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {
/* result might be undefined when input_num is zero */
simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+ // if the system supports SVE or CSSC, __builtin_popcountll
+ // might be compiled to fewer single instructions. For CSSC,
+ // __builtin_popcountll is compiled to a single instruction.
+ return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
}
@@ -9013,15 +14239,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace arm64
@@ -9285,6 +14502,7 @@ namespace {
return vget_lane_u64(
vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
}
+ // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
};
@@ -9933,6 +15151,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -9944,6 +15165,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -9980,6 +15223,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace arm64
@@ -10070,7 +15378,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -10249,6 +15557,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -10288,6 +15597,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -10544,6 +15865,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -10581,6 +16115,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -10599,6 +16143,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -10615,26 +16181,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -10723,7 +16336,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -10806,15 +16419,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -10845,7 +16460,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -10894,7 +16523,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -10993,7 +16622,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -11091,7 +16720,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -11146,7 +16775,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -11232,7 +16861,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -11272,11 +16901,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -11287,9 +16925,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -11338,6 +16975,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -11490,11 +17218,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -11505,9 +17247,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -11556,6 +17297,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -11749,6 +17583,9 @@ public:
#endif // SIMDJSON_ARM64_IMPLEMENTATION_H
/* end file simdjson/arm64/implementation.h */
+// defining SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER allows us to provide our own bit_indexer::write
+#define SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER
+
/* including simdjson/arm64/begin.h: #include <simdjson/arm64/begin.h> */
/* begin file simdjson/arm64/begin.h */
/* defining SIMDJSON_IMPLEMENTATION to "arm64" */
@@ -11861,7 +17698,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {
/* result might be undefined when input_num is zero */
simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+ // if the system supports SVE or CSSC, __builtin_popcountll
+ // might be compiled to fewer single instructions. For CSSC,
+ // __builtin_popcountll is compiled to a single instruction.
+ return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
}
@@ -11898,15 +17742,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace arm64
@@ -12170,6 +18005,7 @@ namespace {
return vget_lane_u64(
vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
}
+ // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
};
@@ -13615,6 +19451,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace arm64
@@ -14048,7 +20174,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -14076,6 +20201,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -14158,7 +20352,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -14455,7 +20648,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -14472,6 +20665,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -14480,6 +20674,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -14528,7 +20723,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -14543,7 +20738,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -14555,8 +20750,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -14573,29 +20768,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -14623,16 +20818,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -14660,11 +20855,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -14704,7 +20917,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -14726,7 +20952,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -14912,9 +21151,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -14937,6 +21180,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -14991,73 +21305,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for arm64 */
-/* including generic/stage2/structural_iterator.h for arm64: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for arm64 */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace arm64 {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace arm64
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for arm64 */
/* including generic/stage2/tape_builder.h for arm64: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for arm64 */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -15080,12 +21327,8 @@ namespace arm64 {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -15138,88 +21381,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -15228,27 +21513,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -15256,7 +21562,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -15270,76 +21577,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -15352,13 +21725,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -15379,6 +21754,38 @@ simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
/* end file generic/stage2/tape_builder.h for arm64 */
/* end file generic/stage2/amalgamated.h for arm64 */
+#undef SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER
+
+namespace simdjson { namespace arm64 { namespace { namespace stage1 {
+
+// The generic bit_indexer::write emits the structural indexes in groups of
+// four (up to 24), so a block with five set bits computes and stores eight
+// indexes, three of them wasted. Typical JSON has four to eight structural
+// characters per 64-byte block. This version writes the first four indexes
+// unconditionally and then continues by groups of two, which cuts the wasted
+// work on such blocks at the cost of one more branch for dense blocks. Both
+// versions fall back to the same scalar loop past 24 indexes.
+simdjson_inline void bit_indexer::write(uint32_t idx, uint64_t bits) {
+ if (bits == 0) { return; }
+
+ const int cnt = static_cast<int>(count_ones(bits));
+#if SIMDJSON_PREFER_REVERSE_BITS
+ bits = reverse_bits(bits);
+#endif
+ write_indexes<0, 4>(idx, bits);
+ if (simdjson_unlikely(4 < cnt)) {
+ write_indexes_stepped<4, 24, 2>(idx, bits, cnt);
+ }
+ if (simdjson_unlikely(24 < cnt)) {
+ for (int i = 24; i < cnt; ++i) {
+ write_index(idx, bits, i);
+ }
+ }
+ this->tail += cnt;
+}
+
+}}}} // namespace simdjson::arm64::(anonymous)::stage1
+
//
// Stage 1
//
@@ -15404,55 +21811,43 @@ namespace {
using namespace simd;
simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) {
- // Functional programming causes trouble with Visual Studio.
- // Keeping this version in comments since it is much nicer:
- // auto v = in.map<uint8_t>([&](simd8<uint8_t> chunk) {
- // auto nib_lo = chunk & 0xf;
- // auto nib_hi = chunk.shr<4>();
- // auto shuf_lo = nib_lo.lookup_16<uint8_t>(16, 0, 0, 0, 0, 0, 0, 0, 0, 8, 12, 1, 2, 9, 0, 0);
- // auto shuf_hi = nib_hi.lookup_16<uint8_t>(8, 0, 18, 4, 0, 1, 0, 1, 0, 0, 0, 3, 2, 1, 0, 0);
- // return shuf_lo & shuf_hi;
- // });
- const simd8<uint8_t> table1(16, 0, 0, 0, 0, 0, 0, 0, 0, 8, 12, 1, 2, 9, 0, 0);
- const simd8<uint8_t> table2(8, 0, 18, 4, 0, 1, 0, 1, 0, 0, 0, 3, 2, 1, 0, 0);
-
- simd8x64<uint8_t> v(
- (in.chunks[0] & 0xf).lookup_16(table1) & (in.chunks[0].shr<4>()).lookup_16(table2),
- (in.chunks[1] & 0xf).lookup_16(table1) & (in.chunks[1].shr<4>()).lookup_16(table2),
- (in.chunks[2] & 0xf).lookup_16(table1) & (in.chunks[2].shr<4>()).lookup_16(table2),
- (in.chunks[3] & 0xf).lookup_16(table1) & (in.chunks[3].shr<4>()).lookup_16(table2)
+ const uint8x16_t op_table = simd8<uint8_t>(
+ 0xff, 0, ',', ':', 0, '[', ']', '{', '}', 0, 0, 0, 0, 0, 0, 0
+ );
+ const uint8x16_t ws_table = simd8<uint8_t>(
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0xff, 0xff, 0, 0, 0xff, 0, 0
);
+ const uint8x16_t d0_0 = in.chunks[0];
+ const uint8x16_t d0_1 = in.chunks[1];
+ const uint8x16_t d0_2 = in.chunks[2];
+ const uint8x16_t d0_3 = in.chunks[3];
- // We compute whitespace and op separately. If the code later only use one or the
- // other, given the fact that all functions are aggressively inlined, we can
- // hope that useless computations will be omitted. This is namely case when
- // minifying (we only need whitespace). *However* if we only need spaces,
- // it is likely that we will still compute 'v' above with two lookup_16: one
- // could do it a bit cheaper. This is in contrast with the x64 implementations
- // where we can, efficiently, do the white space and structural matching
- // separately. One reason for this difference is that on ARM NEON, the table
- // lookups either zero or leave unchanged the characters exceeding 0xF whereas
- // on x64, the equivalent instruction (pshufb) automatically applies a mask,
- // ignoring the 4 most significant bits. Thus the x64 implementation is
- // optimized differently. This being said, if you use this code strictly
- // just for minification (or just to identify the structural characters),
- // there is a small untaken optimization opportunity here. We deliberately
- // do not pick it up.
+ const uint8x16_t match_op_0 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_0, vdupq_n_u8(3)), 4)), d0_0);
+ const uint8x16_t match_op_1 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_1, vdupq_n_u8(3)), 4)), d0_1);
+ const uint8x16_t match_op_2 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_2, vdupq_n_u8(3)), 4)), d0_2);
+ const uint8x16_t match_op_3 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_3, vdupq_n_u8(3)), 4)), d0_3);
- uint64_t op = simd8x64<bool>(
- v.chunks[0].any_bits_set(0x7),
- v.chunks[1].any_bits_set(0x7),
- v.chunks[2].any_bits_set(0x7),
- v.chunks[3].any_bits_set(0x7)
- ).to_bitmask();
+ const uint8x16_t match_ws_0 = vqtbx1q_u8(vceqq_u8(d0_0, vdupq_n_u8(' ')), ws_table, d0_0);
+ const uint8x16_t match_ws_1 = vqtbx1q_u8(vceqq_u8(d0_1, vdupq_n_u8(' ')), ws_table, d0_1);
+ const uint8x16_t match_ws_2 = vqtbx1q_u8(vceqq_u8(d0_2, vdupq_n_u8(' ')), ws_table, d0_2);
+ const uint8x16_t match_ws_3 = vqtbx1q_u8(vceqq_u8(d0_3, vdupq_n_u8(' ')), ws_table, d0_3);
- uint64_t whitespace = simd8x64<bool>(
- v.chunks[0].any_bits_set(0x18),
- v.chunks[1].any_bits_set(0x18),
- v.chunks[2].any_bits_set(0x18),
- v.chunks[3].any_bits_set(0x18)
- ).to_bitmask();
+ const uint8x16_t bit_mask = simd8<uint8_t>(
+ 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80,
+ 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80
+ );
+
+ uint8x16_t op_sum0 = vpaddq_u8(vandq_u8(match_op_0, bit_mask), vandq_u8(match_op_1, bit_mask));
+ uint8x16_t ws_sum0 = vpaddq_u8(vandq_u8(match_ws_0, bit_mask), vandq_u8(match_ws_1, bit_mask));
+ uint8x16_t op_sum1 = vpaddq_u8(vandq_u8(match_op_2, bit_mask), vandq_u8(match_op_3, bit_mask));
+ uint8x16_t ws_sum1 = vpaddq_u8(vandq_u8(match_ws_2, bit_mask), vandq_u8(match_ws_3, bit_mask));
+ op_sum0 = vpaddq_u8(op_sum0, op_sum1);
+ ws_sum0 = vpaddq_u8(ws_sum0, ws_sum1);
+ op_sum0 = vpaddq_u8(op_sum0, op_sum0);
+ ws_sum0 = vpaddq_u8(ws_sum0, ws_sum0);
+ const uint64_t op = vgetq_lane_u64(vreinterpretq_u64_u8(op_sum0), 0);
+ const uint64_t whitespace = vgetq_lane_u64(vreinterpretq_u64_u8(ws_sum0), 0);
return { whitespace, op };
}
@@ -15498,7 +21893,7 @@ simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_
return arm64::stage1::json_minifier::minify<64>(buf, len, dst, dst_len);
}
-simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
+simdjson_flatten simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
this->buf = _buf;
this->len = _len;
return arm64::stage1::json_structural_indexer::index<64>(buf, len, *this, streaming);
@@ -15721,16 +22116,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace haswell
@@ -16483,6 +22868,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -16494,6 +22882,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -16530,6 +22940,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace haswell
@@ -16620,7 +23095,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -16799,6 +23274,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -16838,6 +23314,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -17094,6 +23582,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -17131,6 +23832,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -17149,6 +23860,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -17165,26 +23898,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -17273,7 +24053,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -17356,15 +24136,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -17395,7 +24177,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -17444,7 +24240,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -17543,7 +24339,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -17641,7 +24437,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -17696,7 +24492,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -17782,7 +24578,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -17822,11 +24618,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -17837,9 +24642,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -17888,6 +24692,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -18040,11 +24935,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -18055,9 +24964,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18106,6 +25014,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -18465,16 +25466,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace haswell
@@ -20024,6 +27015,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace haswell
@@ -20457,7 +27738,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -20485,6 +27765,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -20567,7 +27916,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -20864,7 +28212,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -20881,6 +28229,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -20889,6 +28238,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -20937,7 +28287,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -20952,7 +28302,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -20964,8 +28314,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -20982,29 +28332,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -21032,16 +28382,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -21069,11 +28419,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -21113,7 +28481,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -21135,7 +28516,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -21321,9 +28715,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -21346,6 +28744,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -21400,73 +28869,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for haswell */
-/* including generic/stage2/structural_iterator.h for haswell: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for haswell */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace haswell {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace haswell
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for haswell */
/* including generic/stage2/tape_builder.h for haswell: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for haswell */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -21489,12 +28891,8 @@ namespace haswell {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -21547,88 +28945,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -21637,27 +29077,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -21665,7 +29126,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -21679,76 +29141,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -21761,13 +29289,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -22123,16 +29653,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace icelake
@@ -22888,6 +30408,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -22899,6 +30422,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -22935,6 +30480,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace icelake
@@ -23025,7 +30635,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -23204,6 +30814,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -23243,6 +30854,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -23499,6 +31122,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -23536,6 +31372,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -23554,6 +31400,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -23570,26 +31438,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -23678,7 +31593,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -23761,15 +31676,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -23800,7 +31717,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -23849,7 +31780,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -23948,7 +31879,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -24046,7 +31977,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -24101,7 +32032,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -24187,7 +32118,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -24227,9 +32158,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
p += parse_digit(*p, i);
bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
// no integer digits, or 0123 (zero must be solo)
if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -24239,12 +32408,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
int64_t exponent = 0;
bool overflow;
- if (simdjson_likely(*p == '.')) {
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
p++;
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24276,7 +32537,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
exponent += exp_neg ? 0-exp : exp;
}
- if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+ if (*p != '"') { return NUMBER_ERROR; }
overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
@@ -24293,163 +32554,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
//
// Check for minus sign
//
- bool negative = (*src == '-');
- src += uint8_t(negative);
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
//
// Parse the integer part.
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
}
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
-
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
+ return INCORRECT_TYPE;
}
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
- }
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -24460,9 +32597,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24501,9 +32637,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
// Assemble (or slow-parse) the float
//
- double d;
+ float d;
if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
}
if (!parse_float_fallback(src - uint8_t(negative), &d)) {
return NUMBER_ERROR;
@@ -24866,16 +33002,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace icelake
@@ -26428,6 +34554,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace icelake
@@ -26861,7 +35277,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -26889,6 +35304,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -26971,7 +35455,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -27268,7 +35751,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -27285,6 +35768,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -27293,6 +35777,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -27341,7 +35826,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -27356,7 +35841,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -27368,8 +35853,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -27386,29 +35871,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -27436,16 +35921,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -27473,11 +35958,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -27517,7 +36020,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -27539,7 +36055,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -27725,9 +36254,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -27750,6 +36283,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -27804,73 +36408,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for icelake */
-/* including generic/stage2/structural_iterator.h for icelake: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for icelake */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace icelake {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace icelake
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for icelake */
/* including generic/stage2/tape_builder.h for icelake: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for icelake */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -27893,12 +36430,8 @@ namespace icelake {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -27951,88 +36484,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -28041,27 +36616,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -28069,7 +36665,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -28083,76 +36680,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -28165,13 +36828,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -28542,16 +37207,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace ppc64
@@ -29450,6 +38105,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -29461,6 +38119,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -29497,6 +38177,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace ppc64
@@ -29587,7 +38332,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -29766,6 +38511,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -29805,6 +38551,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -30061,6 +38819,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -30098,6 +39069,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -30116,6 +39097,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -30132,26 +39135,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -30240,7 +39290,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -30323,15 +39373,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -30362,7 +39414,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -30411,7 +39477,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -30510,7 +39576,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -30608,7 +39674,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -30663,7 +39729,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -30749,7 +39815,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -30789,11 +39855,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -30804,9 +39879,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -30855,6 +39929,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -31007,11 +40172,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -31022,9 +40201,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -31073,6 +40251,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -31398,16 +40669,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace ppc64
@@ -33103,6 +42364,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace ppc64
@@ -33536,7 +43087,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -33564,6 +43114,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -33646,7 +43265,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -33943,7 +43561,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -33960,6 +43578,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -33968,6 +43587,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -34016,7 +43636,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -34031,7 +43651,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -34043,8 +43663,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -34061,29 +43681,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -34111,16 +43731,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -34148,11 +43768,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -34192,7 +43830,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -34214,7 +43865,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -34400,9 +44064,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -34425,6 +44093,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -34479,73 +44218,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for ppc64 */
-/* including generic/stage2/structural_iterator.h for ppc64: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for ppc64 */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace ppc64 {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace ppc64
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for ppc64 */
/* including generic/stage2/tape_builder.h for ppc64: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for ppc64 */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -34568,12 +44240,8 @@ namespace ppc64 {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -34626,88 +44294,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -34716,27 +44426,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -34744,7 +44475,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -34758,76 +44490,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -34840,13 +44638,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -35161,16 +44961,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -35749,16 +45539,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -36372,6 +46152,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -36383,6 +46166,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -36419,6 +46224,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace westmere
@@ -36509,7 +46379,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -36688,6 +46558,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -36727,6 +46598,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -36983,6 +46866,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -37020,6 +47116,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -37038,6 +47144,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -37054,26 +47182,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -37162,7 +47337,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -37245,15 +47420,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -37284,7 +47461,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -37333,7 +47524,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -37432,7 +47623,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37530,7 +47721,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37585,7 +47776,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37671,7 +47862,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37711,9 +47902,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
p += parse_digit(*p, i);
bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
// no integer digits, or 0123 (zero must be solo)
if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -37723,12 +48152,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
int64_t exponent = 0;
bool overflow;
- if (simdjson_likely(*p == '.')) {
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
p++;
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37760,7 +48281,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
exponent += exp_neg ? 0-exp : exp;
}
- if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+ if (*p != '"') { return NUMBER_ERROR; }
overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
@@ -37777,163 +48298,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
//
// Check for minus sign
//
- bool negative = (*src == '-');
- src += uint8_t(negative);
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
//
// Parse the integer part.
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
}
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
+ return INCORRECT_TYPE;
}
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -37944,9 +48341,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37985,9 +48381,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
// Assemble (or slow-parse) the float
//
- double d;
+ float d;
if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
}
if (!parse_float_fallback(src - uint8_t(negative), &d)) {
return NUMBER_ERROR;
@@ -38332,16 +48728,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -38920,16 +49306,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -40340,6 +50716,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace westmere
@@ -40773,7 +51439,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -40801,6 +51466,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -40883,7 +51617,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -41180,7 +51913,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -41197,6 +51930,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -41205,6 +51939,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -41253,7 +51988,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -41268,7 +52003,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -41280,8 +52015,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -41298,29 +52033,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -41348,16 +52083,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -41385,11 +52120,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -41429,7 +52182,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -41451,7 +52217,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -41637,9 +52416,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -41662,6 +52445,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -41716,73 +52570,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for westmere */
-/* including generic/stage2/structural_iterator.h for westmere: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for westmere */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace westmere {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace westmere
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for westmere */
/* including generic/stage2/tape_builder.h for westmere: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for westmere */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -41805,12 +52592,8 @@ namespace westmere {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -41863,88 +52646,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -41953,27 +52778,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -41981,7 +52827,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -41995,76 +52842,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -42077,13 +52990,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -42390,10 +53305,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lasx
@@ -43140,6 +54051,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -43151,6 +54065,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -43187,6 +54123,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace lasx
@@ -43277,7 +54278,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -43456,6 +54457,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -43495,6 +54497,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -43751,6 +54765,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -43788,6 +55015,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -43806,6 +55043,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -43822,26 +55081,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -43930,7 +55236,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -44013,15 +55319,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -44052,7 +55360,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -44101,7 +55423,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -44200,7 +55522,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -44298,7 +55620,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -44353,7 +55675,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -44439,7 +55761,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -44479,11 +55801,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -44494,9 +55825,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -44545,6 +55875,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -44697,11 +56118,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -44712,9 +56147,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -44763,6 +56197,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -45061,10 +56588,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lasx
@@ -46608,6 +58131,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace lasx
@@ -47041,7 +58854,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -47069,6 +58881,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -47151,7 +59032,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -47448,7 +59328,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -47465,6 +59345,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -47473,6 +59354,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -47521,7 +59403,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -47536,7 +59418,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -47548,8 +59430,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -47566,29 +59448,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -47616,16 +59498,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -47653,11 +59535,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -47697,7 +59597,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -47719,7 +59632,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -47905,9 +59831,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -47930,6 +59860,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -47984,73 +59985,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for lasx */
-/* including generic/stage2/structural_iterator.h for lasx: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for lasx */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace lasx {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace lasx
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for lasx */
/* including generic/stage2/tape_builder.h for lasx: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for lasx */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -48073,12 +60007,8 @@ namespace lasx {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -48131,88 +60061,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -48221,27 +60193,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -48249,7 +60242,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -48263,76 +60257,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -48345,13 +60405,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -48613,10 +60675,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lsx
@@ -49345,6 +61403,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -49356,6 +61417,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -49392,6 +61475,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace lsx
@@ -49482,7 +61630,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -49661,6 +61809,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -49700,6 +61849,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -49956,6 +62117,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -49993,6 +62367,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -50011,6 +62395,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -50027,26 +62433,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -50135,7 +62588,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -50218,15 +62671,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -50257,7 +62712,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -50306,7 +62775,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -50405,7 +62874,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -50503,7 +62972,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -50558,7 +63027,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -50644,7 +63113,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -50684,9 +63153,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
p += parse_digit(*p, i);
bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
// no integer digits, or 0123 (zero must be solo)
if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -50696,12 +63403,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
int64_t exponent = 0;
bool overflow;
- if (simdjson_likely(*p == '.')) {
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
p++;
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -50733,7 +63532,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
exponent += exp_neg ? 0-exp : exp;
}
- if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+ if (*p != '"') { return NUMBER_ERROR; }
overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
@@ -50750,163 +63549,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
//
// Check for minus sign
//
- bool negative = (*src == '-');
- src += uint8_t(negative);
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
//
// Parse the integer part.
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
}
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
- exponent += exp_neg ? 0-exp : exp;
+ return INCORRECT_TYPE;
}
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
-
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
- }
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -50917,9 +63592,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -50958,9 +63632,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
// Assemble (or slow-parse) the float
//
- double d;
+ float d;
if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
}
if (!parse_float_fallback(src - uint8_t(negative), &d)) {
return NUMBER_ERROR;
@@ -51251,10 +63925,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lsx
@@ -52780,6 +65450,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace lsx
@@ -53213,7 +66173,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -53241,6 +66200,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -53323,7 +66351,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -53620,7 +66647,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -53637,6 +66664,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -53645,6 +66673,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -53693,7 +66722,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -53708,7 +66737,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -53720,8 +66749,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -53738,29 +66767,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -53788,16 +66817,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -53825,11 +66854,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -53869,7 +66916,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -53891,7 +66951,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -54077,9 +67150,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -54102,6 +67179,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -54156,73 +67304,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for lsx */
-/* including generic/stage2/structural_iterator.h for lsx: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for lsx */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace lsx {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace lsx
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for lsx */
/* including generic/stage2/tape_builder.h for lsx: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for lsx */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -54245,12 +67326,8 @@ namespace lsx {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -54303,88 +67380,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -54393,27 +67512,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -54421,7 +67561,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -54435,76 +67576,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -54517,13 +67724,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -54793,11 +68002,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
return __builtin_popcountll(input_num);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace rvv_vls
@@ -55538,6 +68742,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -55549,6 +68756,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -55585,6 +68814,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace rvv_vls
@@ -55675,7 +68969,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -55854,6 +69148,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -55893,6 +69188,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -56149,6 +69456,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -56186,6 +69706,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -56204,6 +69734,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -56220,26 +69772,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -56328,7 +69927,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -56411,15 +70010,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -56450,7 +70051,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -56499,7 +70114,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -56598,7 +70213,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -56696,7 +70311,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -56751,7 +70366,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -56837,7 +70452,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -56877,11 +70492,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -56892,9 +70516,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -56943,6 +70566,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -57095,11 +70809,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -57110,9 +70838,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -57161,6 +70888,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -57339,7 +71159,7 @@ public:
simdjson_inline implementation() : simdjson::implementation(
"rvv_vls",
"RISC-V V extension",
- 0
+ internal::instruction_set::RVV_VLS
) {}
simdjson_warn_unused error_code create_dom_parser_implementation(
size_t capacity,
@@ -57456,11 +71276,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
return __builtin_popcountll(input_num);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace rvv_vls
@@ -59371,6 +73186,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace rvv_vls
@@ -59804,7 +73909,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
return EMPTY;
}
}
-
parser.n_structural_indexes = new_structural_indexes;
} else if (partial == stage1_mode::streaming_final) {
if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -59832,6 +73936,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
// the trailing garbage.
return EMPTY;
}
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(have_unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(have_unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
}
checker.check_eof();
return checker.errors();
@@ -59914,7 +74087,6 @@ namespace {
namespace stage2 {
class json_iterator;
-class structural_iterator;
struct tape_builder;
struct tape_writer;
@@ -60211,7 +74383,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -60228,6 +74400,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -60236,6 +74409,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -60284,7 +74458,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -60299,7 +74473,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -60311,8 +74485,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -60329,29 +74503,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -60379,16 +74553,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -60416,11 +74590,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -60460,7 +74652,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -60482,7 +74687,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -60668,9 +74886,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -60693,6 +74915,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -60747,73 +75040,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t
#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
/* end file generic/stage2/stringparsing.h for rvv_vls */
-/* including generic/stage2/structural_iterator.h for rvv_vls: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for rvv_vls */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace rvv_vls {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
- const uint8_t* const buf;
- uint32_t *next_structural;
- dom_parser_implementation &dom_parser;
-
- // Start a structural
- simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
- : buf{_dom_parser.buf},
- next_structural{&_dom_parser.structural_indexes[start_structural_index]},
- dom_parser{_dom_parser} {
- }
- // Get the buffer position of the current structural character
- simdjson_inline const uint8_t* current() {
- return &buf[*(next_structural-1)];
- }
- // Get the current structural character
- simdjson_inline char current_char() {
- return buf[*(next_structural-1)];
- }
- // Get the next structural character without advancing
- simdjson_inline char peek_next_char() {
- return buf[*next_structural];
- }
- simdjson_inline const uint8_t* peek() {
- return &buf[*next_structural];
- }
- simdjson_inline const uint8_t* advance() {
- return &buf[*(next_structural++)];
- }
- simdjson_inline char advance_char() {
- return buf[*(next_structural++)];
- }
- simdjson_inline size_t remaining_len() {
- return dom_parser.len - *(next_structural-1);
- }
-
- simdjson_inline bool at_end() {
- return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
- }
- simdjson_inline bool at_beginning() {
- return next_structural == dom_parser.structural_indexes.get();
- }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace rvv_vls
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for rvv_vls */
/* including generic/stage2/tape_builder.h for rvv_vls: #include <generic/stage2/tape_builder.h> */
/* begin file generic/stage2/tape_builder.h for rvv_vls */
#ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -60836,12 +75062,8 @@ namespace rvv_vls {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -60894,88 +75116,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -60984,27 +75248,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -61012,7 +75297,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -61026,76 +75312,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -61108,13 +75460,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -61708,6 +76062,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -61719,6 +76076,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -61755,6 +76134,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace fallback
@@ -61845,7 +76289,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -62024,6 +76468,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -62063,6 +76508,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -62319,6 +76776,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -62356,6 +77026,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -62374,6 +77054,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -62390,26 +77092,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -62498,7 +77247,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -62581,15 +77330,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -62620,7 +77371,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -62669,7 +77434,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -62768,7 +77533,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -62866,7 +77631,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -62921,7 +77686,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -63007,7 +77772,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -63047,9 +77812,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
p += parse_digit(*p, i);
bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
// no integer digits, or 0123 (zero must be solo)
if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -63059,12 +78062,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
int64_t exponent = 0;
bool overflow;
- if (simdjson_likely(*p == '.')) {
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
p++;
- while (parse_digit(*p, i)) { p++; }
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -63096,7 +78191,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
exponent += exp_neg ? 0-exp : exp;
}
- if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+ if (*p != '"') { return NUMBER_ERROR; }
overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
@@ -63113,163 +78208,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
//
// Check for minus sign
//
- bool negative = (*src == '-');
- src += uint8_t(negative);
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
//
// Parse the integer part.
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
}
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
-
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
+ return INCORRECT_TYPE;
}
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -63280,9 +78251,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -63321,9 +78291,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
// Assemble (or slow-parse) the float
//
- double d;
+ float d;
if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
}
if (!parse_float_fallback(src - uint8_t(negative), &d)) {
return NUMBER_ERROR;
@@ -63871,6 +78841,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
return 0;
}
+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ * unless stage 1 dropped a trailing unclosed string, whose bytes it
+ * never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start,
+ size_t scan_len) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Phase 1: Scan structural_indexes to find RS positions and handle them.
+ // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+ // is already in structural_indexes (it's an operator). For scalars like numbers,
+ // the digit following RS is NOT in structural_indexes because the scanner sees
+ // RS as a scalar, making the digit a scalar continuation, not a start.
+ // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+ // actual value start position.
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_rs_pos = 0;
+ uint32_t rs_count = 0;
+
+ for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+ const uint32_t pos = parser.structural_indexes[read_idx];
+ if (parser.buf[pos] == 0x1E) {
+ // This is an RS character - find the actual JSON value start.
+ last_rs_pos = pos;
+ rs_count++;
+ // Skip past this RS and any whitespace *and any additional RSes*
+ // to locate the real value. Consecutive RSes are degenerate
+ // "empty records" per RFC 7464; we collapse them here. They do
+ // not always appear as separate entries in structural_indexes
+ // because the scanner groups runs of adjacent non-whitespace
+ // scalar bytes (including RS) into a single scalar start.
+ uint32_t value_start = pos + 1;
+ while (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+ value_start++;
+ } else if (c == 0x1E) {
+ // Collapsed empty record. Still count it so rs_count reflects
+ // the true number of record markers and last_rs_pos tracks
+ // the final one.
+ last_rs_pos = value_start;
+ rs_count++;
+ value_start++;
+ } else {
+ break;
+ }
+ }
+ // If the scanner emitted additional structurals inside the
+ // whitespace+RS run we just walked over (i.e., isolated RSes
+ // separated by whitespace), skip past them so we do not
+ // double-count or double-emit.
+ while (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] < value_start) {
+ read_idx++;
+ }
+ // Check if the value start is an operator (always present in
+ // scanner structural_indexes) or a scalar-like start (which may
+ // be missing from structural_indexes and must be added here).
+ // Note: '"' is NOT always in structural_indexes. The scanner
+ // classifies '"' as a scalar character and emits it as a
+ // structural only when it is a *scalar start* (preceded by
+ // whitespace or an operator). When '"' immediately follows an
+ // RS (which the scanner also classifies as scalar), it is
+ // treated as a scalar continuation and not emitted - so we
+ // must add it here just like any other scalar value.
+ if (value_start < scan_len) {
+ const uint8_t c = parser.buf[value_start];
+ const bool is_operator =
+ (c == '{' || c == '}' || c == '[' || c == ']' ||
+ c == ':' || c == ',');
+ // If the next scanner structural is exactly at value_start,
+ // the scanner already emitted it (it followed whitespace) and
+ // we must not add a duplicate - a subsequent iteration will
+ // copy it into write_idx.
+ const bool already_emitted =
+ (read_idx + 1 < parser.n_structural_indexes &&
+ parser.structural_indexes[read_idx + 1] == value_start);
+ if (!is_operator && !already_emitted) {
+ // Scalar value (number/true/false/null/string) - add its
+ // position since scanner missed it.
+ parser.structural_indexes[write_idx++] = value_start;
+ }
+ }
+ } else {
+ // Not RS, copy to output
+ parser.structural_indexes[write_idx++] = pos;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) {
+ // Only RS markers here: the last one opens a record continuing past the
+ // window, so that is where the next batch resumes.
+ if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+ return 0;
+ }
+ if (rs_count == 0) {
+ // No RS found; for final batch, try generic boundary detection
+ return is_final ? find_next_document_index(parser) : 0;
+ }
+
+ // Phase 2: Determine batch boundaries based on RS positions
+
+ if (is_final) {
+ // Final batch: all documents are complete (last one ends at EOF).
+ // In json_sequence mode, RS markers define document boundaries, so all
+ // remaining structurals form complete documents. Return them all directly.
+ // (Calling find_next_document_index() would fail for scalar documents.)
+ return parser.n_structural_indexes;
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document starting at an RS is complete if there is another RS after it.
+ next_batch_start = last_rs_pos;
+
+ if (rs_count < 2) {
+ // Only one RS, so we have at most one document that may be incomplete.
+ // We cannot confirm it is complete without another RS.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+ // Find the structural index cutoff: keep only structurals < last_rs_pos.
+ // Since we already filtered RS, all remaining structurals are valid.
+ // We iterate backward to find the last structural before last_rs_pos.
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_rs_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ // No structurals before the last RS - no complete documents
+ if (keep_count == 0) { return 0; }
+
+ // All documents before the last RS are complete by definition (the next RS
+ // confirms their end). No need to call find_next_document_index() which
+ // would fail for scalar documents like `1` or `"hello"`.
+ return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ * 0 if no document content found (EMPTY),
+ * or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+ dom_parser_implementation &parser,
+ size_t len,
+ bool is_final,
+ uint32_t &next_batch_start) {
+ // Default: next batch starts at end of buffer
+ next_batch_start = uint32_t(len);
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ // Track depth to identify root-level commas (depth 0)
+ int depth = 0;
+ // The EOF sentinel: len, or where a discarded unclosed string starts.
+ const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+ uint32_t write_idx = 0;
+ uint32_t last_root_comma_pos = 0;
+ uint32_t root_comma_count = 0;
+
+ for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+ uint32_t idx = parser.structural_indexes[i];
+ uint8_t c = parser.buf[idx];
+
+ switch (c) {
+ case '{': case '[':
+ depth++;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case '}': case ']':
+ depth--;
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ case ',':
+ if (depth == 0) {
+ // Root-level comma = document boundary, skip it
+ last_root_comma_pos = idx;
+ root_comma_count++;
+ continue;
+ }
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ default:
+ // Colons, scalars, etc.
+ parser.structural_indexes[write_idx++] = idx;
+ break;
+ }
+ }
+
+ // Update structural index count. Compaction left a stale index in the slot
+ // past the end: restore the EOF sentinel that stage 1 had planted there, which
+ // document_stream::truncated_bytes() reads after a final batch.
+ parser.n_structural_indexes = write_idx;
+ parser.structural_indexes[write_idx] = sentinel;
+
+ if (parser.n_structural_indexes == 0) { return 0; }
+
+ if (is_final) {
+ // Final batch: use standard boundary detection on filtered indexes
+ return find_next_document_index(parser);
+ }
+
+ // Partial batch: need to find complete documents only.
+ // A document ending with a root comma is complete.
+ if (root_comma_count == 0) {
+ // No root commas found; we cannot confirm any document is complete.
+ // The whole batch might be one incomplete document.
+ // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+ return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+ }
+
+ // We have at least one root comma. Documents before the last comma are complete.
+ next_batch_start = last_root_comma_pos + 1;
+
+ // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+ uint32_t keep_count = 0;
+ for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+ if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+ keep_count = i;
+ break;
+ }
+ }
+
+ if (keep_count == 0) { return 0; }
+
+ // Use standard boundary detection on the complete portion
+ parser.n_structural_indexes = keep_count;
+ return find_next_document_index(parser);
+}
+
} // namespace stage1
} // unnamed namespace
} // namespace fallback
@@ -64050,9 +79310,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
within the unicode codepoint handling code. */
src += bs_dist;
dst += bs_dist;
- if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
- return nullptr;
- }
+ // Decode adjacent Unicode escapes without returning to the
+ // quote-and-backslash scanner between code points.
+ do {
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } while (src[0] == '\\' && src[1] == 'u');
} else {
/* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
* write bs_dist+1 characters to output
@@ -64075,6 +79339,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
}
}
+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+ // Far from the end: identical to parse_string's loop. The guard uses
+ // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+ // buf_end; escape/Unicode look-aheads read within the string (before the
+ // closing quote, which is < buf_end), so they are in bounds here too.
+ // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+ // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+ // following bytes (+6 from the '\'), then (if it sees \u) another
+ // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+ // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+ // chunk start. The +12 margin ensures that even on kernels where
+ // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+ // in-bounds. The scratch fallback (3*PAD) is already safe.
+ while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+ auto b = backslash_and_quote{};
+ auto bs_quote = b.copy_and_find(src, dst);
+ if (bs_quote.has_quote_first()) {
+ return dst + bs_quote.quote_index();
+ }
+ if (bs_quote.has_backslash()) {
+ auto bs_dist = bs_quote.backslash_index();
+ uint8_t escape_char = src[bs_dist + 1];
+ if (escape_char == 'u') {
+ src += bs_dist;
+ dst += bs_dist;
+ if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+ return nullptr;
+ }
+ } else {
+ uint8_t escape_result = escape_map[escape_char];
+ if (escape_result == 0u) {
+ return nullptr;
+ }
+ dst[bs_dist] = escape_result;
+ src += bs_dist + 2;
+ dst += bs_dist + 1;
+ }
+ } else {
+ src += backslash_and_quote::BYTES_PROCESSED;
+ dst += backslash_and_quote::BYTES_PROCESSED;
+ }
+ }
+ // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+ // space-padded scratch (spaces are neither quote nor backslash, so they do not
+ // disturb matching) and let the regular parser finish from there. The closing
+ // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+ // the chunk starting at some offset <= remaining and reads at most
+ // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+ // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+ // extra bytes past an escape).
+ uint8_t scratch[SIMDJSON_PADDING * 3];
+ const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+ std::memset(scratch, ' ', sizeof(scratch));
+ std::memcpy(scratch, src, remaining);
+ return parse_string(scratch, dst, allow_replacement);
+}
+
simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
// It is not ideal that this function is nearly identical to parse_string.
while (1) {
@@ -64278,7 +79613,7 @@ public:
*
* - increment_count(iter) - each time a value is found in an array or object.
*/
- template<bool STREAMING, typename V>
+ template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;
/**
@@ -64295,6 +79630,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *peek() const noexcept;
/**
* Advance to the next token.
@@ -64303,6 +79639,7 @@ public:
*
* They may include invalid JSON as well (such as `1.2.3` or `ture`).
*/
+ template<bool UNPADDED>
simdjson_inline const uint8_t *advance() noexcept;
/**
* Get the remaining length of the document, from the start of the current token.
@@ -64351,7 +79688,7 @@ public:
simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
};
-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
logger::log_start();
@@ -64366,7 +79703,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
// Read first value
//
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
// Make sure the outer object or array is closed before continuing; otherwise, there are ways we
// could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -64378,8 +79715,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
}
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
}
}
@@ -64396,29 +79733,29 @@ object_begin:
SIMDJSON_TRY( visitor.visit_object_start(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.increment_count(*this) );
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
object_field:
- if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+ if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
object_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',':
SIMDJSON_TRY( visitor.increment_count(*this) );
{
- auto key = advance();
+ auto key = advance<UNPADDED>();
if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
SIMDJSON_TRY( visitor.visit_key(*this, key) );
}
@@ -64446,16 +79783,16 @@ array_begin:
array_value:
{
- auto value = advance();
+ auto value = advance<UNPADDED>();
switch (*value) {
- case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
- case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+ case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+ case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
}
}
array_continue:
- switch (*advance()) {
+ switch (*advance<UNPADDED>()) {
case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -64483,11 +79820,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
dom_parser{_dom_parser} {
}
+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
- return &buf[*(next_structural)];
+ const uint32_t idx = *(next_structural);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
+template<bool UNPADDED>
simdjson_inline const uint8_t *json_iterator::advance() noexcept {
- return &buf[*(next_structural++)];
+ const uint32_t idx = *(next_structural++);
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ if (simdjson_unlikely(idx >= dom_parser.len)) {
+ static constexpr uint8_t unpadded_eof_sentinel = 0;
+ return &unpadded_eof_sentinel;
+ }
+ }
+ return &buf[idx];
}
simdjson_inline size_t json_iterator::remaining_len() const noexcept {
return dom_parser.len - *(next_structural-1);
@@ -64527,7 +79882,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
case '"': return visitor.visit_root_string(*this, value);
case 't': return visitor.visit_root_true_atom(*this, value);
case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_root_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_root_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
case '-':
case '0': case '1': case '2': case '3': case '4':
case '5': case '6': case '7': case '8': case '9':
@@ -64549,7 +79917,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
switch (*value) {
case 't': return visitor.visit_true_atom(*this, value);
case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+ case 'n': {
+ auto err = visitor.visit_null_atom(*this, value);
+ if (err == SUCCESS) { return err; }
+ // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+ return visitor.visit_nan_atom(*this, value, err);
+ }
+ // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+ case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+ case 'i':
+ case 'I': return visitor.visit_inf_atom(*this, value);
+#else
case 'n': return visitor.visit_null_atom(*this, value);
+#endif
default:
log_error("Non-value found when value was expected!");
return TAPE_ERROR;
@@ -64720,12 +80101,8 @@ namespace fallback {
namespace {
namespace stage2 {
-struct tape_builder {
- template<bool STREAMING>
- simdjson_warn_unused static simdjson_inline error_code parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
/** Called when a non-empty document starts. */
simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
/** Called when a non-empty document ends without error. */
@@ -64778,88 +80155,130 @@ struct tape_builder {
simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#if SIMDJSON_ENABLE_NAN_INF
+ simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+ // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+ // this returns a tape error on failure.
+ simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+ simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
/** Called each time a new field or element in an array or object is found. */
simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;
/** Next location to write to tape */
tape_writer tape;
+public:
+ simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
private:
/** Next write location in the string buf for stage 2 parsing */
uint8_t *current_string_buf_loc;
- simdjson_inline tape_builder(dom::document &doc) noexcept;
-
simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
simdjson_inline void start_container(json_iterator &iter) noexcept;
simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl
-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
- dom_parser_implementation &dom_parser,
- dom::document &doc) noexcept {
- dom_parser.doc = &doc;
- json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
- tape_builder builder(doc);
- return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+ template<bool STREAMING>
+ simdjson_warn_unused static simdjson_inline error_code parse_document(
+ dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+ dom_parser.doc = &doc;
+ json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+ if (dom_parser._unpadded) {
+ tape_builder_impl<true> builder(doc);
+ return iter.walk_document<STREAMING, true>(builder);
+ } else {
+ tape_builder_impl<false> builder(doc);
+ return iter.walk_document<STREAMING, false>(builder);
+ }
+ }
+};
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_root_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
return iter.visit_primitive(*this, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
start_container(iter);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
constexpr uint32_t start_tape_index = 0;
tape.append(start_tape_index, internal::tape_type::ROOT);
tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
return visit_string(iter, key, true);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
return SUCCESS;
}
-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
iter.log_value(key ? "key" : "string");
uint8_t *dst = on_start_string(iter);
- dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+ // We do not allow replacement when the escape characters are invalid.
+ // UNPADDED is a compile-time constant chosen once per document by
+ // tape_builder::parse_document, so the padded build instantiates only the
+ // plain parse_string call below -- no runtime branch and no flag load.
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+ } else {
+ dst = stringparsing::parse_string(value+1, dst, false);
+ }
if (dst == nullptr) {
iter.log_error("Invalid escape in string");
return STRING_ERROR;
@@ -64868,27 +80287,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
return visit_string(iter, value);
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("number");
- error_code err = numberparsing::parse_number(value, tape);
+ const uint8_t *num = value;
+ std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+ SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+ // numberparsing reads ahead in 8-byte blocks for floats
+ // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+ // number whose digits reach the final bytes of an unpadded buffer would read
+ // past it. *(next_structural) is the offset of the token following this
+ // number, hence an upper bound on where the digits end; when that is within
+ // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+ // (mirroring visit_root_number). This fires only for numbers near the end.
+ if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+ const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+ copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+ if (copy.get() == nullptr) { return MEMALLOC; }
+ std::memcpy(copy.get(), value, rl);
+ std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+ num = copy.get();
+ }
+ }
+ error_code err = numberparsing::parse_number(num, tape);
if (simdjson_unlikely(err == BIGINT_ERROR &&
iter.dom_parser._number_as_string)) {
// Write big integer to string buffer using the same format as strings.
// Scan digits the same way parse_number does (skip optional '-', then digits).
- const uint8_t *p = value;
+ const uint8_t *p = num;
if (*p == '-') p++;
while (numberparsing::is_digit(*p)) p++;
// The digit run must be terminated by a structural or whitespace character; otherwise the
// token is malformed (e.g. "123456789123456789123x").
if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
- size_t len = size_t(p - value);
+ size_t len = size_t(p - num);
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
- memcpy(dst, value, len);
+ memcpy(dst, num, len);
dst += len;
on_end_string(dst);
return SUCCESS;
@@ -64896,7 +80336,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
return err;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
//
// We need to make a copy to make sure that the string is space terminated.
// This is not about padding the input, which should already padded up
@@ -64910,76 +80351,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
// practice unless you are in the strange scenario where you have many JSON
// documents made of single atoms.
//
- std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+ // In a stream, the input goes on with other documents: copy up to the next
+ // structural only, not to the end of the batch.
+ const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+ std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
if (copy.get() == nullptr) { return MEMALLOC; }
- std::memcpy(copy.get(), value, iter.remaining_len());
- std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+ std::memcpy(copy.get(), value, len);
+ std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
error_code error = visit_number(iter, copy.get());
return error;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
- if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+ // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+ // token at the very end of an unpadded buffer would over-read. Use the
+ // length-aware form there (the root variant already does this).
+ const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_true_atom(value);
+ if (!ok) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("true");
if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
tape.append(0, internal::tape_type::TRUE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
- if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_false_atom(value);
+ if (!ok) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("false");
if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
tape.append(0, internal::tape_type::FALSE_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
- if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+ const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_null_atom(value);
+ if (!ok) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
iter.log_value("null");
if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
tape.append(0, internal::tape_type::NULL_VALUE);
return SUCCESS;
}
+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ // For unpadded input use the length-aware validator so the 'infinity'-style
+ // 8-byte compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_nan_atom(value);
+ if (!ok) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+ iter.log_value("nan");
+ if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+ tape.append_double(std::numeric_limits<double>::quiet_NaN());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+ // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+ // compare cannot read past the buffer on a malformed token at the end.
+ const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+ : atomparsing::is_valid_inf_atom(value);
+ if (!ok) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+ iter.log_value("inf");
+ // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+ if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+ tape.append_double(std::numeric_limits<double>::infinity());
+ return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
// private:
-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
auto start_index = next_tape_index(iter);
tape.append(start_index+2, start);
tape.append(start_index, end);
return SUCCESS;
}
-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
iter.dom_parser.open_containers[iter.depth].count = 0;
tape.skip(); // We don't actually *write* the start element until the end.
}
-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
// Write the ending tape element, pointing at the start location
const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
tape.append(start_tape_index, end);
@@ -64992,13 +80499,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
return SUCCESS;
}
-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
// we advance the point, accounting for the fact that we have a NULL termination
tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
return current_string_buf_loc + sizeof(uint32_t);
}
-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
// TODO check for overflow in case someone has a crazy string (>=4GB?)
// But only add the overflow check when the document itself exceeds 4GB
@@ -65229,14 +80738,14 @@ simdjson_warn_unused simdjson_inline error_code scan() {
// Primitive or invalid character (invalid characters will be checked in stage 2)
} else {
// Anything else, add the structural and go until we find the next one.
- // We also stop on '"' so that an unclosed string still reaches
- // validate_string(); a quote swallowed by the run would hide it. A
- // quote cannot occur inside a valid primitive. We deliberately do not
- // stop on every ESC_ASCII character: that also covers a backslash and the
- // control characters, and ending the run there makes the fallback
- // disagree with the SIMD kernels.
+ // We also stop on RS (0x1E) so that RFC 7464 json_sequence inputs
+ // like `\x1e"a"\x1e"b"` produce a separate structural for each RS
+ // rather than being absorbed into a single primitive run, and on '"'
+ // so that an unclosed string still reaches validate_string(). Neither
+ // can occur inside a valid primitive.
add_structural();
- while (idx+1<len && !char_is_space_or_operator(buf[idx+1]) && buf[idx+1] != '"') {
+ while (idx+1<len && !char_is_space_or_operator(buf[idx+1]) &&
+ buf[idx+1] != 0x1e && buf[idx+1] != '"') {
idx++;
};
}
@@ -65291,6 +80800,75 @@ simdjson_warn_unused simdjson_inline error_code scan() {
// doing.
parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
if (parser.n_structural_indexes == 0) { return EMPTY; }
+ } else if (partial == stage1_mode::json_sequence_partial) {
+ // RFC 7464: use RS positions for batch boundaries
+ // A discarded unclosed string also caps how far the filter may scan.
+ size_t scan_len = len;
+ if(unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return CAPACITY; }
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::json_sequence_final) {
+ // RFC 7464: final batch, last document extends to EOF
+ // As above: a discarded unclosed string caps how far the filter may scan.
+ size_t scan_len = len;
+ if(unclosed_string) {
+ parser.n_structural_indexes--;
+ scan_len = parser.structural_indexes[parser.n_structural_indexes];
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+ // The filter compacted structural_indexes in place and restored the EOF
+ // sentinel past the compacted end, so the copy below is either the start
+ // of a truncated document or len, as in streaming_final.
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return EMPTY; }
+ } else if (partial == stage1_mode::comma_delimited_partial) {
+ // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+ if(unclosed_string) {
+ parser.n_structural_indexes--;
+ if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return CAPACITY; }
+ }
+ uint32_t next_batch_start = uint32_t(len);
+ auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+ if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+ return CAPACITY;
+ }
+ if (new_structural_indexes == 0) {
+ // An EMPTY batch must still advance next_batch_start, or document_stream
+ // re-parses the same bytes forever. CAPACITY when it cannot.
+ if (next_batch_start == 0) { return CAPACITY; }
+ parser.n_structural_indexes = 0;
+ parser.structural_indexes[0] = next_batch_start;
+ return EMPTY;
+ }
+ parser.n_structural_indexes = new_structural_indexes;
+ parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+ } else if (partial == stage1_mode::comma_delimited_final) {
+ // Comma-delimited: final batch, last document extends to EOF
+ if(unclosed_string) { parser.n_structural_indexes--; }
+ uint32_t next_batch_start = uint32_t(len);
+ parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+ parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+ parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+ if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return EMPTY; }
} else if(unclosed_string) { error = UNCLOSED_STRING; }
return error;
}
diff --git a/deps/simdjson/simdjson.h b/deps/simdjson/simdjson.h
index 43fe09631c0..58041ff68a7 100644
--- a/deps/simdjson/simdjson.h
+++ b/deps/simdjson/simdjson.h
@@ -1,4 +1,4 @@
-/* auto-generated on 2026-09-04 16:04:31 -0400. version 4.6.11 Do not edit! */
+/* auto-generated on 2026-10-07 22:43:29 -0400. version 5.0.3 Do not edit! */
/* including simdjson.h: */
/* begin file simdjson.h */
#ifndef SIMDJSON_H
@@ -61,7 +61,9 @@
#endif
// C++ 26
-#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202402L) // update when the standard is finalized
+// While C++26 is a working draft, compilers report 202400L in C++26 mode
+// (both GCC 16 and Clang 21 do). Update when the standard is finalized.
+#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202400L)
#define SIMDJSON_CPLUSPLUS26 1
#endif
@@ -118,14 +120,48 @@
#endif
#endif
-// The current specification is unclear on how we detect
-// static reflection, both __cpp_lib_reflection and
-// __cpp_impl_reflection are proposed in the draft specification.
-// For now, we disable static reflect by default. It must be
-// specified at compiler time.
+// Static reflection.
+//
+// The reflection-based APIs (simdjson::to, document::get<T>, the builder,
+// compile-time JSON, annotations) need considerably more than the reflection
+// operator. We turn them on only when the compiler advertises all of:
+//
+// P2996 reflection (^^, splicers, <meta>) __cpp_impl_reflection,
+// __cpp_lib_reflection
+// P1306 expansion statements (template for) __cpp_expansion_statements
+// P3491 std::define_static_string / _array __cpp_lib_define_static
+//
+// Two further features we rely on have, as of this writing, no feature-test
+// macro of their own, so they cannot be checked directly:
+//
+// P3394 annotations ([[=x]], std::meta::annotations_of) -- used for
+// the annotations of simdjson/annotations.h (rename, skip, ...).
+// P3289 consteval blocks (consteval { ... }) -- used by compile_time_json.
+//
+// Every implementation that defines the four macros above also implements
+// those two, so requiring the four is sufficient in practice. If that ever
+// stops being true, define SIMDJSON_STATIC_REFLECTION=0 to opt out.
+//
+// SIMDJSON_STATIC_REFLECTION may always be defined by the user (or by the
+// build system) to 0 or 1 to override the detection.
+//
+// Note that C++26 mode alone is not enough: GCC 16 requires -freflection,
+// and only then does it define __cpp_impl_reflection.
#ifndef SIMDJSON_STATIC_REFLECTION
-#define SIMDJSON_STATIC_REFLECTION 0 // disabled by default.
+#if defined(SIMDJSON_CPLUSPLUS26) && \
+ defined(__cpp_impl_reflection) && __cpp_impl_reflection >= 202506L && \
+ defined(__cpp_lib_reflection) && __cpp_lib_reflection >= 202506L && \
+ defined(__cpp_expansion_statements) && \
+ __cpp_expansion_statements >= 202506L && \
+ defined(__cpp_lib_define_static) && __cpp_lib_define_static >= 202506L
+// __cpp_lib_reflection is the feature-test macro for <meta>, so there is no
+// need for a separate __has_include check (which would have to be guarded for
+// compilers that lack __has_include).
+#define SIMDJSON_STATIC_REFLECTION 1
+#else
+#define SIMDJSON_STATIC_REFLECTION 0
#endif
+#endif // SIMDJSON_STATIC_REFLECTION
#if defined(__apple_build_version__)
#if __apple_build_version__ < 14000000
@@ -158,6 +194,47 @@
#define SIMDJSON_SUPPORTS_DESERIALIZATION 0
#endif
+// The C++20 char8_t type (and std::u8string/std::u8string_view) is available.
+// Because all strings that simdjson produces are valid UTF-8, we can offer
+// char8_t variants of our string accessors when this macro is set.
+#if !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+#if defined(__cpp_char8_t) && __cpp_char8_t >= 201811L
+#define SIMDJSON_SUPPORTS_CHAR8_T 1
+#else
+#define SIMDJSON_SUPPORTS_CHAR8_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+
+// The C++23 fixed-width floating-point types std::float32_t and std::float64_t
+// (<stdfloat>) are available. They are optional even in C++23: a compiler that
+// provides them predefines __STDCPP_FLOAT32_T__ and __STDCPP_FLOAT64_T__.
+// When these macros are set, we offer get_float32() and get_float64().
+#if !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+#if defined(__STDCPP_FLOAT32_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT32_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT32_T
+#define SIMDJSON_SUPPORTS_FLOAT32_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+
+#if !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+#if defined(__STDCPP_FLOAT64_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT64_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT64_T
+#define SIMDJSON_SUPPORTS_FLOAT64_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T || SIMDJSON_SUPPORTS_FLOAT64_T
+#include <stdfloat>
+#endif
+
#if !defined(SIMDJSON_CONSTEVAL)
#if defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
@@ -166,6 +243,18 @@
#define SIMDJSON_CONSTEVAL 0
#endif // defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
#endif // !defined(SIMDJSON_CONSTEVAL)
+
+// SIMDJSON_CONSTEXPR_STRING is 'constexpr' when the standard library supports
+// constexpr std::string (e.g., libstdc++ 12 or better), and empty otherwise. It
+// lets functions that build a std::string be constant expressions when possible
+// while still compiling against older standard libraries.
+#if !defined(SIMDJSON_CONSTEXPR_STRING)
+#if defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#define SIMDJSON_CONSTEXPR_STRING constexpr
+#else
+#define SIMDJSON_CONSTEXPR_STRING
+#endif // defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#endif // !defined(SIMDJSON_CONSTEXPR_STRING)
#endif // SIMDJSON_COMPILER_CHECK_H
/* end file simdjson/compiler_check.h */
/* including simdjson/portability.h: #include "simdjson/portability.h" */
@@ -457,16 +546,86 @@ using std::size_t;
#endif
#endif
+#ifndef SIMDJSON_HAS_UNISTD_H
+#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+#define SIMDJSON_HAS_UNISTD_H 1
+#else
+#define SIMDJSON_HAS_UNISTD_H 0
+#endif
+#endif
+
+// padded_memory_map availability.
+//
+// On POSIX platforms the class is always available: the implementation uses
+// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
+//
+// On Windows the class is disabled by default and must be explicitly
+// opted into by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. Enabling
+// it requires:
+// 1. `<windows.h>` has been included *before* `<simdjson.h>` (so that
+// this header can see the Win32 types and the `_WINDOWS_` include
+// guard),
+// 2. the compilation targets Windows 10, version 1803 or later
+// (i.e. `NTDDI_VERSION >= NTDDI_WIN10_RS4`, `0x0A000005`). This is
+// required because the implementation relies on the modern memory
+// APIs introduced with that version (`CreateFileMapping2` /
+// `MapViewOfFile3`),
+// 3. the link step pulls in an import library that exports those APIs,
+// typically `onecore.lib` (or `mincore.lib`).
+//
+// The `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS` CMake option arranges (1)-(3)
+// automatically when building simdjson with its own CMake. Consumers using
+// simdjson as a pre-built library are responsible for setting the macro,
+// the Windows version macros, and the link library themselves.
+//
+// If the opt-in conditions are not met on Windows, `padded_memory_map`
+// simply does not exist -- any attempt to use it fails at compile time
+// with an "unknown identifier" diagnostic rather than silently degrading.
+//
+// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
+// available in the current translation unit. Users may test this macro to
+// conditionally compile code that depends on padded_memory_map.
+#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
+ #if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+ #elif defined(_WINDOWS_) && defined(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS) && SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+ #else
+ #define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
+ #endif
+#endif
#endif // SIMDJSON_PORTABILITY_H
/* end file simdjson/portability.h */
+#include <cstddef>
namespace simdjson {
namespace internal {
+/**
+ * @private
+ * Scratch capacity that every caller of to_chars must provide.
+ *
+ * The emitted decimal is at most ~24 characters, but dragonbox() and
+ * format_buffer() intentionally write past the logical end with fixed-size
+ * 16/17-byte memcpy/memset operations so the compiler can inline them (no
+ * libc mem* dispatch with size-class branches). The extra bytes are required
+ * for safety of those over-writes; do not shrink this below 40.
+ * See src/to_chars.cpp and #2805.
+ */
+// Use an unscoped enum (not static constexpr / inline constexpr):
+// - C++11 targets (readme_examples11, quickstart11, ...) still include this header
+// - a static constexpr in the amalgamated simdjson.cpp TU is unused there
+// (only callers in headers use it) and trips -Wunused-const-variable -Werror
+enum : size_t { to_chars_buffer_size = 40 };
/**
* @private
* Our own implementation of the C++17 to_chars function.
* Defined in src/to_chars
+ *
+ * @note The buffer starting at first must have at least to_chars_buffer_size
+ * bytes of writable storage (see to_chars_buffer_size).
+ * @note The input number must be finite (NaN/Inf are not supported).
+ * @note The result is NOT null-terminated.
*/
char *to_chars(char *first, const char *last, double value);
/**
@@ -476,6 +635,12 @@ char *to_chars(char *first, const char *last, double value);
*/
double from_chars(const char *first) noexcept;
double from_chars(const char *first, const char* end) noexcept;
+/**
+ * @private
+ * Same as from_chars, but produces a correctly rounded binary32 (float) value.
+ * Defined in src/from_chars
+ */
+float from_chars_float(const char *first) noexcept;
}
#ifndef SIMDJSON_EXCEPTIONS
@@ -486,6 +651,10 @@ double from_chars(const char *first, const char* end) noexcept;
#endif
#endif
+#ifndef SIMDJSON_ENABLE_NAN_INF
+#define SIMDJSON_ENABLE_NAN_INF 0
+#endif
+
} // namespace simdjson
#if defined(__GNUC__)
@@ -501,16 +670,14 @@ double from_chars(const char *first, const char* end) noexcept;
// Align to N-byte boundary
#define SIMDJSON_ROUNDUP_N(a, n) (((a) + ((n)-1)) & ~((n)-1))
-#define SIMDJSON_ROUNDDOWN_N(a, n) ((a) & ~((n)-1))
-
-#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
#if SIMDJSON_REGULAR_VISUAL_STUDIO
// We could use [[deprecated]] but it requires C++14
#define simdjson_deprecated __declspec(deprecated)
#define simdjson_really_inline __forceinline
- #define simdjson_never_inline __declspec(noinline)
+ #define simdjson_never_inline inline __declspec(noinline)
+ #define simdjson_really_flatten [[msvc::flatten]]
#define simdjson_unused
#define simdjson_warn_unused
@@ -551,6 +718,7 @@ double from_chars(const char *first, const char* end) noexcept;
#define simdjson_really_inline inline __attribute__((always_inline))
#define simdjson_never_inline inline __attribute__((noinline))
+ #define simdjson_really_flatten [[gnu::flatten]]
#define simdjson_unused __attribute__((unused))
#define simdjson_warn_unused __attribute__((warn_unused_result))
@@ -627,6 +795,15 @@ double from_chars(const char *first, const char* end) noexcept;
#define simdjson_inline simdjson_really_inline
#endif
+#if defined(simdjson_flatten)
+ // Prefer the user's definition of simdjson_flatten; don't define it ourselves.
+#elif (defined(__GNUC__) && !defined(__OPTIMIZE__)) || (defined(_DEBUG) && _MSC_VER )
+ // Flattening can lead to significant code bloat and high compile times. Don't use it for unoptimized builds.
+ #define simdjson_flatten
+#else
+ #define simdjson_flatten simdjson_really_flatten
+#endif
+
#if SIMDJSON_VISUAL_STUDIO
/**
* Windows users need to do some extra work when building
@@ -2538,22 +2715,22 @@ namespace std {
#define SIMDJSON_SIMDJSON_VERSION_H
/** The version of simdjson being used (major.minor.revision) */
-#define SIMDJSON_VERSION "4.6.11"
+#define SIMDJSON_VERSION "5.0.3"
namespace simdjson {
enum {
/**
* The major version (MAJOR.minor.revision) of simdjson being used.
*/
- SIMDJSON_VERSION_MAJOR = 4,
+ SIMDJSON_VERSION_MAJOR = 5,
/**
* The minor version (major.MINOR.revision) of simdjson being used.
*/
- SIMDJSON_VERSION_MINOR = 6,
+ SIMDJSON_VERSION_MINOR = 0,
/**
* The revision (major.minor.REVISION) of simdjson being used.
*/
- SIMDJSON_VERSION_REVISION = 11
+ SIMDJSON_VERSION_REVISION = 3
};
} // namespace simdjson
@@ -2625,6 +2802,7 @@ enum error_code {
OUT_OF_BOUNDS, ///< Attempted to access location outside of document.
TRAILING_CONTENT, ///< Unexpected trailing content in the JSON input
OUT_OF_CAPACITY, ///< The capacity was exceeded, we cannot allocate enough memory.
+ UNKNOWN_FIELD, ///< JSON field does not map to any member of the target (see simdjson::deny_unknown_fields)
NUM_ERROR_CODES ///< Placeholder for end of error code list.
};
@@ -2987,6 +3165,7 @@ inline const std::string error_message(int error) noexcept;
#if SIMDJSON_SUPPORTS_CONCEPTS
#include <concepts>
+#include <string_view>
#include <type_traits>
namespace simdjson {
@@ -3028,6 +3207,19 @@ concept constructible_from_string_view = std::is_constructible_v<T, std::string_
&& !std::is_same_v<T, std::string_view>
&& std::is_default_constructible_v<T>;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * A C++20 char8_t string type such as std::u8string. Such types cannot be built
+ * from a std::string_view (the character types differ), so they need their own
+ * deserialization path, going through the u8 string accessors.
+ */
+template<typename T>
+concept constructible_from_u8string_view = std::is_constructible_v<T, std::u8string_view>
+ && !std::is_same_v<T, std::u8string_view>
+ && !std::is_constructible_v<T, std::string_view>
+ && std::is_default_constructible_v<T>;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename M>
concept string_view_keyed_map = string_view_like<typename M::key_type>
&& requires(std::remove_cvref_t<M>& m, typename M::key_type sv, typename M::mapped_type v) {
@@ -3122,9 +3314,15 @@ concept string_like =
// Concept that checks if a type is a container but not a string (because
// strings handling must be handled differently)
// Now uses iterator-based approach for broader container support
+//
+// Optional types are excluded on purpose. Since C++26 (P3168), std::optional
+// is itself a range, so without the exclusion an std::optional would match
+// both this concept and optional_type, making the container and the optional
+// overloads of atom()/append() ambiguous. See issue 2827.
template <typename T>
concept container_but_not_string =
- std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>;
+ std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::optional_type<T>;
@@ -3231,6 +3429,11 @@ struct fixed_string {
data[i] = str[i];
}
}
+ constexpr fixed_string(const unsigned char (&str)[N]) {
+ for (std::size_t i = 0; i < N; ++i) {
+ data[i] = static_cast<char>(str[i]);
+ }
+ }
char data[N];
constexpr std::string_view view() const { return {data, N - 1}; }
constexpr size_t size() const { return N ; }
@@ -3262,6 +3465,11 @@ struct string_constant {
#endif // SIMDJSON_CONSTEVALUTIL_H
/* end file simdjson/constevalutil.h */
+#if SIMDJSON_SUPPORTS_CHAR8_T
+#include <string>
+#include <string_view>
+#endif
+
/**
* @brief The top level simdjson namespace, containing everything the library provides.
*/
@@ -3271,6 +3479,10 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS
/** The maximum document size supported by simdjson. */
constexpr size_t SIMDJSON_MAXSIZE_BYTES = 0xFFFFFFFF;
+/** The maximum depth of nested objects and arrays supported by simdjson.
+ A depth of SIMDJSON_MAXSIZE_BYTES/2 is not reasonable and would be
+ adversarial, but it serves as an upper bound for validation purposes. */
+constexpr size_t SIMDJSON_MAX_DEPTH = SIMDJSON_MAXSIZE_BYTES/2;
/**
* The amount of padding needed in a buffer to parse JSON.
@@ -3296,6 +3508,30 @@ struct padded_string;
class padded_string_view;
enum class stage1_mode;
+/**
+ * Stream format for parse_many/iterate_many.
+ */
+enum class stream_format {
+ whitespace_delimited, ///< Whitespace-delimited JSON documents (default, includes NDJSON/JSONL)
+ json_sequence, ///< RFC 7464 JSON text sequences (RS-delimited)
+ comma_delimited, ///< Comma-separated JSON documents (e.g., `{...},{...},{...}`)
+ comma_delimited_array,///< A single JSON array whose elements are iterated as
+ ///< comma-separated documents (e.g., `[{...},{...},{...}]`).
+ ///< The parser strips the outer `[` / `]` plus any
+ ///< surrounding JSON whitespace (space, tab, LF, CR)
+ ///< and then behaves like `comma_delimited` over the
+ ///< remaining bytes.
+ newline_delimited ///< NDJSON/JSON Lines where each document occupies exactly
+ ///< one line: documents are separated by line feeds and no
+ ///< document contains a raw line feed. Same inputs as
+ ///< `whitespace_delimited`, but the stronger guarantee lets
+ ///< the parser find the end of a document without walking
+ ///< it. On ondemand `iterate_many`, an unread remainder may
+ ///< be skipped by jumping to the next line feed without
+ ///< structure-validating that remainder. Use
+ ///< `whitespace_delimited` if unsure.
+};
+
namespace internal {
template<typename T>
@@ -3306,6 +3542,52 @@ class tape_ref;
struct value128;
enum class tape_type;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Reinterpret a UTF-8 string as a C++20 std::u8string_view. No byte is copied
+ * or modified: char8_t and char have the same size, representation and
+ * alignment. Every string that simdjson produces is valid UTF-8, so this is a
+ * lossless view over the very same memory.
+ * @private
+ */
+simdjson_inline std::u8string_view as_u8string_view(std::string_view v) noexcept {
+ return std::u8string_view(reinterpret_cast<const char8_t *>(v.data()), v.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
+/**
+ * Assign a UTF-8 string to a string-like receiver. The general case simply
+ * assigns the std::string_view: it covers std::string and any user type that
+ * can be assigned from a std::string_view.
+ * @private
+ */
+template <typename string_type>
+simdjson_inline void assign_utf8(string_type &receiver, std::string_view content) noexcept {
+ receiver = content;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Assign a UTF-8 string to a char8_t-based string (e.g., std::u8string). This
+ * overload is more specialized than the general one, so overload resolution
+ * prefers it whenever the receiver holds char8_t.
+ * @private
+ */
+template <typename traits_type, typename allocator_type>
+simdjson_inline void assign_utf8(std::basic_string<char8_t, traits_type, allocator_type> &receiver, std::string_view content) noexcept {
+ receiver.assign(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+
+/**
+ * Assign a UTF-8 string to a char8_t-based string view (e.g., std::u8string_view).
+ * @private
+ */
+template <typename traits_type>
+simdjson_inline void assign_utf8(std::basic_string_view<char8_t, traits_type> &receiver, std::string_view content) noexcept {
+ receiver = std::basic_string_view<char8_t, traits_type>(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
} // namespace internal
} // namespace simdjson
@@ -3599,7 +3881,12 @@ class document;
* 3) The stream_final mode allows us to truncate final
* unterminated strings. It is useful in conjunction with streaming_partial.
*/
-enum class stage1_mode { regular, streaming_partial, streaming_final};
+enum class stage1_mode {
+ regular,
+ streaming_partial, streaming_final,
+ json_sequence_partial, json_sequence_final,
+ comma_delimited_partial, comma_delimited_final
+};
/**
* Returns true if mode == streaming_partial or mode == streaming_final
@@ -3611,7 +3898,6 @@ inline bool is_streaming(stage1_mode mode) {
// return (mode == stage1_mode::streaming_partial || mode == stage1_mode::streaming_final);
}
-
namespace internal {
@@ -3796,6 +4082,16 @@ public:
/** Whether to store big integers as strings instead of returning BIGINT_ERROR */
bool _number_as_string{false};
+ /**
+ * Whether the input buffer passed to parse() is *not* padded to len +
+ * SIMDJSON_PADDING bytes. When true, stage 2 string parsing avoids reading
+ * past buf+len (it finishes the final, near-the-end bytes from a small padded
+ * scratch buffer). This is set only by the no-padding DOM parse entry points
+ * (dom::parser::parse_unpadded); the default padded fast path leaves it false
+ * and is unaffected.
+ */
+ bool _unpadded{false};
+
protected:
// Declaring these so that subclasses can use them to implement their constructors.
@@ -4350,11 +4646,26 @@ inline std::ostream& operator<<(std::ostream& out, const padded_string& s) { ret
inline std::ostream& operator<<(std::ostream& out, simdjson_result<padded_string> &s) noexcept(false) { return out << s.value(); }
#endif
-
-#ifndef _WIN32
+#if SIMDJSON_HAS_PADDED_MEMORY_MAP
/**
* A class representing a memory-mapped file with padding.
- * It is only available on non-Windows platforms, as Windows has different APIs for memory mapping.
+ *
+ * On POSIX systems (Linux, macOS, BSD, ...), this uses `mmap` to map the file
+ * contents directly into memory, which is efficient for large files (no copy).
+ *
+ * On Windows, this class is disabled by default and must be opted into at
+ * build time by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. When
+ * enabled, `<windows.h>` must also be included before `<simdjson.h>` and
+ * the compilation must target Windows 10, version 1803 or later. The
+ * Windows implementation uses the modern memory APIs (`VirtualAlloc2`,
+ * `CreateFileMapping2`, `MapViewOfFile3`) with the placeholder virtual
+ * memory mechanism to always achieve true zero-copy mapping with
+ * contiguous zero-filled padding.
+ *
+ * Either way, the resulting `padded_string_view` carries at least
+ * `SIMDJSON_PADDING` bytes of accessible zero-filled padding after the file
+ * content, so it can be consumed directly by the simdjson parsers (including
+ * `parse_many` / `iterate_many`).
*/
class padded_memory_map {
public:
@@ -4362,9 +4673,11 @@ public:
* Create a new padded memory map for the given file.
* After creating the memory map, you can call view() to get a padded_string_view of the file content.
* The memory map will be automatically released when the padded_memory_map instance is destroyed.
- * Note that the file content is not copied, so this is efficient for large files. However,
- * the file must remain unchanged while the memory map is in use. In case of error (e.g., file not found,
- * permission denied, etc.), the memory map will be invalid and view() will return an empty view.
+ * On POSIX systems, the file content is not copied, so this is efficient for large files.
+ * On Windows, the file is mapped into memory via `MapViewOfFile3` (zero-copy).
+ * In all cases, the file must remain unchanged while the memory map is in use.
+ * In case of error (e.g., file not found, permission denied, etc.), the memory map will be
+ * invalid and view() will return an empty view.
* You can check if the memory map is valid by calling is_valid() before using view().
*
* @param filename the path to the file to memory-map.
@@ -4401,8 +4714,14 @@ private:
padded_memory_map &operator=(const padded_memory_map &) = delete;
const char *data{nullptr};
size_t size{0};
+#ifdef _WIN32
+ // When the file ends near an allocation-granularity boundary, we use the
+ // placeholder API to append zero-filled padding pages. This pointer tracks
+ // that region so the destructor can release it with VirtualFree.
+ void *padding_view_{nullptr};
+#endif
};
-#endif // _WIN32
+#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
@@ -4476,6 +4795,9 @@ simdjson_warn_unused error_code minify(const char *buf, size_t len, char *dst, s
#include <memory>
#include <string>
#include <ostream>
+#if SIMDJSON_CPLUSPLUS17
+#include <variant>
+#endif
namespace simdjson {
@@ -4540,6 +4862,68 @@ public:
}; // padded_string_view
+/**
+ * Get the system's memory page size. By default, we return
+ * 4096 bytes, which is the most common page size. On systems
+ * where the page size is not a multiple of 4096 bytes, and not
+ * a unix-like system, nor Windows, this function may return an
+ * incorrect value.
+ *
+ * @return The page size in bytes.
+ */
+inline uint32_t get_page_size() noexcept;
+
+#if SIMDJSON_CPLUSPLUS17
+/**
+ * A padded_input is a wrapper around either a padded_string_view or a padded_string.
+ * It will automatically pad a string_view if it does not have sufficient padding
+ * up to the end of the memory page. Note that a requirement for this method to
+ * make sense is to be on a system with a page size of at least 4096 (which is
+ * universal except on some embedded systems).
+ */
+struct padded_input {
+ /**
+ * Construct a padded_input from a string_view. If the string_view does not have sufficient padding,
+ * the data will be copied into a padded_string and the padded_string_view will point to the
+ * padded_string's data. Otherwise, the padded_string_view will point to the original string_view's data.
+ */
+ inline explicit padded_input(std::string_view sv);
+ /**
+ * Construct a padded_input from a C-style string (length specified). If the string does not have sufficient padding,
+ * the data will be copied into a padded_string and the padded_string_view will point to the
+ * padded_string's data. Otherwise, the padded_string_view will point to the original string's data.
+ */
+ inline explicit padded_input(const char *data, size_t length);
+ /**
+ * Construct a padded_input from a std::string. If the string does not have sufficient padding
+ * (considering its capacity), the data will be copied into a padded_string and the padded_string_view
+ * will point to the padded_string's data. Otherwise, the padded_string_view will point to the
+ * original string's data.
+ */
+ inline explicit padded_input(const std::string &s);
+
+ /**
+ * Check if the padded_input is a view.
+ *
+ * @return true if the padded_input is a view, false otherwise.
+ */
+ inline bool is_view() const noexcept;
+
+ /**
+ * Convert the padded_input to a padded_string_view.
+ *
+ * @return The padded_string_view.
+ */
+ inline operator simdjson::padded_string_view() const noexcept;
+
+private:
+ std::variant<simdjson::padded_string_view, simdjson::padded_string> storage;
+ // whether we cross a page boundary and need to allocate a new padded string.
+ static inline bool needs_allocation(const char* buf, size_t len, size_t padding = SIMDJSON_PADDING) noexcept;
+};
+
+#endif // SIMDJSON_CPLUSPLUS17
+
#if SIMDJSON_EXCEPTIONS
/**
* Send padded_string instance to an output stream.
@@ -4572,6 +4956,31 @@ inline padded_string_view pad(std::string& s) noexcept;
* @return The padded string.
*/
inline padded_string_view pad_with_reserve(std::string& s) noexcept;
+
+/**
+ * Return the index-th document-aligned slice of a delimited stream.
+ *
+ * The input is divided into blocks of block_size bytes and each boundary is
+ * moved forward to just past the next delimiter, so a document is never split.
+ * The delimiter must not occur inside a document: a line feed for NDJSON, a
+ * record separator (0x1E) for RFC 7464.
+ *
+ * Slices are contiguous and non-overlapping, and each may be parsed
+ * independently, so callers can process them on as many threads as they like.
+ *
+ * Iterate while index * block_size < data.size(). A slice is empty when its
+ * block falls entirely inside one document, which happens only if that document
+ * is longer than block_size; skip it and continue. With block_size larger than
+ * the longest document, no slice is ever empty.
+ *
+ * @param data The padded input.
+ * @param delimiter The byte that separates documents.
+ * @param block_size The nominal slice size, before snapping.
+ * @param index Which slice to return, counting from zero.
+ */
+inline padded_string_view slice_at(padded_string_view data, char delimiter,
+ size_t block_size, size_t index) noexcept;
+
} // namespace simdjson
#endif // SIMDJSON_PADDED_STRING_VIEW_H
@@ -4588,6 +4997,15 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept;
#include <cstring> /* memcmp */
+// for page size computation.
+#if SIMDJSON_HAS_UNISTD_H
+ #include <unistd.h>
+ #if defined(__APPLE__)
+ #include <sys/sysctl.h>
+ #endif
+#endif
+
+
namespace simdjson {
inline padded_string_view::padded_string_view(const char* s, size_t len, size_t capacity) noexcept
@@ -4677,7 +5095,102 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept {
return padded_string_view(s.data(), s.size(), s.capacity());
}
+inline uint32_t get_page_size() noexcept {
+#if defined(_WINDOWS_) // if and only if someone loaded Windows.h, we can get the page size from there.
+// Otherwise, we assume 4096.
+ static const uint32_t cached = []() -> uint32_t {
+ SYSTEM_INFO si;
+ GetSystemInfo(&si);
+ return static_cast<std::uint32_t>(si.dwPageSize);
+ }();
+ return cached;
+#elif SIMDJSON_HAS_UNISTD_H
+ static const uint32_t cached = []() -> uint32_t {
+ long page_size = sysconf(_SC_PAGESIZE);
+ if (page_size > 0) {
+ return static_cast<uint32_t>(page_size);
+ }
+ return 4096; // fallback
+ }();
+ return cached;
+#else
+ return 4096; // fallback
+#endif
+}
+#if SIMDJSON_CPLUSPLUS17
+
+inline padded_input::padded_input(std::string_view sv)
+ : storage(simdjson::padded_string_view{}) {
+ if (needs_allocation(sv.data(), sv.size())) {
+ storage = simdjson::padded_string(sv);
+ } else {
+ storage = simdjson::padded_string_view(
+ sv.data(), sv.size(), sv.size() + simdjson::SIMDJSON_PADDING);
+ }
+}
+
+inline padded_input::padded_input(const char *data, size_t length)
+ : storage(simdjson::padded_string_view{}) {
+ if (needs_allocation(data, length)) {
+ storage = simdjson::padded_string(data, length);
+ } else {
+ storage = simdjson::padded_string_view(
+ data, length, length + simdjson::SIMDJSON_PADDING);
+ }
+}
+inline padded_input::padded_input(const std::string &s)
+ : storage(simdjson::padded_string_view{}) {
+ const size_t len = s.size();
+ const size_t cap = s.capacity();
+ // Here we have the string content from data() to data() + size(),
+ // but the memory is accessible from data() to data() + capacity().
+ const size_t needed_padding = (cap - len) < simdjson::SIMDJSON_PADDING
+ ? simdjson::SIMDJSON_PADDING - (cap - len) : 0;
+ if (needed_padding > 0 && needs_allocation(s.data(), cap, needed_padding)) {
+ storage = simdjson::padded_string(s);
+ } else {
+ storage = simdjson::padded_string_view(
+ s.data(), len, len + simdjson::SIMDJSON_PADDING);
+ }
+}
+
+inline bool padded_input::is_view() const noexcept {
+ return std::holds_alternative<simdjson::padded_string_view>(storage);
+}
+
+inline padded_input::operator simdjson::padded_string_view() const noexcept {
+ return std::visit([](const auto& p) -> simdjson::padded_string_view {
+ return p;
+ }, storage);
+}
+
+inline bool padded_input::needs_allocation(const char* buf, size_t len, size_t padding) noexcept {
+ if(len == 0) { return false; }
+ const auto page_size = get_page_size();
+ return ((reinterpret_cast<uintptr_t>(buf + len - 1) % page_size)
+ + padding >= static_cast<uintptr_t>(page_size));
+}
+#endif // SIMDJSON_CPLUSPLUS17
+
+inline padded_string_view slice_at(padded_string_view data, char delimiter,
+ size_t block_size, size_t index) noexcept {
+ if (block_size == 0 || index > data.size() / block_size) { return {}; }
+ const size_t raw_begin = index * block_size;
+ if (raw_begin >= data.size()) { return {}; }
+
+ auto snap = [&](size_t want) -> size_t {
+ if (want >= data.size()) { return data.size(); }
+ const void *p = std::memchr(data.data() + want, delimiter, data.size() - want);
+ return p ? size_t(static_cast<const char *>(p) - data.data()) + 1 : data.size();
+ };
+
+ const size_t begin = (raw_begin == 0) ? 0 : snap(raw_begin);
+ const size_t end = snap(raw_begin + block_size);
+ if (begin >= end) { return {}; }
+ return padded_string_view(data.data() + begin, end - begin,
+ data.capacity() - begin);
+}
} // namespace simdjson
@@ -4688,13 +5201,18 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept {
#include <climits>
#include <cwchar>
-#ifndef _WIN32
+#if SIMDJSON_HAS_UNISTD_H
#include <fcntl.h>
#include <stdio.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <unistd.h>
#endif
+// On Windows, `padded_memory_map` (when it is enabled) depends on types and
+// functions declared in <windows.h>. We deliberately do NOT include that
+// header here: users of simdjson who want `padded_memory_map` on Windows
+// must include <windows.h> themselves *before* including this header. See
+// padded_string.h for the detection logic.
namespace simdjson {
namespace internal {
@@ -5064,7 +5582,9 @@ inline bool padded_string_builder::reserve(size_t additional) noexcept {
}
-#ifndef _WIN32
+#if SIMDJSON_HAS_PADDED_MEMORY_MAP
+
+#if SIMDJSON_HAS_UNISTD_H
simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
int fd = open(filename, O_RDONLY);
@@ -5100,7 +5620,132 @@ simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
munmap(const_cast<char *>(data), size + simdjson::SIMDJSON_PADDING);
}
}
+#elif defined(_WIN32)
+// Windows zero-copy implementation using placeholder virtual memory.
+//
+// We use the modern Windows memory APIs (VirtualAlloc2, CreateFileMapping2,
+// MapViewOfFile3 -- available since Windows 10 1803) to map the file into a
+// contiguous virtual address range that includes at least SIMDJSON_PADDING
+// zero bytes after the file content, with no data copies.
+//
+// Strategy:
+// 1. If rounding the file size up to the allocation granularity already
+// exceeds file_size + SIMDJSON_PADDING, the OS page zero-fill provides
+// the padding and we use a simple MapViewOfFile3 call.
+// 2. Otherwise we reserve a contiguous placeholder region via VirtualAlloc2,
+// split it at the granularity-aligned file boundary, map the file into
+// the first part, and commit zero pages for the second part (padding).
+simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
+ HANDLE file_handle = ::CreateFileA(
+ filename, GENERIC_READ,
+ FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
+ NULL, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, NULL);
+ if (file_handle == INVALID_HANDLE_VALUE) {
+ return;
+ }
+ LARGE_INTEGER file_size_li;
+ if (!::GetFileSizeEx(file_handle, &file_size_li) || file_size_li.QuadPart < 0) {
+ ::CloseHandle(file_handle);
+ return;
+ }
+#if SIMDJSON_IS_32BITS
+ if (static_cast<unsigned long long>(file_size_li.QuadPart) >
+ static_cast<unsigned long long>(SIZE_MAX - simdjson::SIMDJSON_PADDING)) {
+ ::CloseHandle(file_handle);
+ return;
+ }
+#endif
+ size = static_cast<size_t>(file_size_li.QuadPart);
+ if (size == 0) {
+ ::CloseHandle(file_handle);
+ return;
+ }
+
+ HANDLE section = ::CreateFileMapping2(
+ file_handle, NULL, FILE_MAP_READ, PAGE_READONLY,
+ 0, 0, NULL, NULL, 0);
+ ::CloseHandle(file_handle);
+ if (section == NULL) {
+ return;
+ }
+
+ SYSTEM_INFO si;
+ ::GetSystemInfo(&si);
+ const size_t granularity = static_cast<size_t>(si.dwAllocationGranularity);
+ const size_t file_region = (size + granularity - 1) & ~(granularity - 1);
+ const size_t total_needed = size + simdjson::SIMDJSON_PADDING;
+
+ if (file_region >= total_needed) {
+ // The zero-fill in the last page already covers the padding.
+ PVOID view = ::MapViewOfFile3(
+ section, ::GetCurrentProcess(), NULL, 0, 0,
+ 0, PAGE_READONLY, NULL, 0);
+ ::CloseHandle(section);
+ if (view != NULL) {
+ data = static_cast<const char *>(view);
+ }
+ return;
+ }
+
+ // We need extra zero pages beyond the file region. Use the placeholder API
+ // to get a contiguous virtual address range spanning both the file mapping
+ // and the zero-filled padding.
+ const size_t padding_region =
+ ((total_needed - file_region) + granularity - 1) & ~(granularity - 1);
+ const size_t reserve_size = file_region + padding_region;
+
+ // Reserve a contiguous placeholder.
+ PVOID placeholder = ::VirtualAlloc2(
+ ::GetCurrentProcess(), NULL, reserve_size,
+ MEM_RESERVE | MEM_RESERVE_PLACEHOLDER, PAGE_NOACCESS, NULL, 0);
+ if (placeholder == NULL) {
+ ::CloseHandle(section);
+ return;
+ }
+ // Split into two placeholders at the file_region boundary.
+ if (!::VirtualFree(placeholder, file_region,
+ MEM_RELEASE | MEM_PRESERVE_PLACEHOLDER)) {
+ ::VirtualFree(placeholder, 0, MEM_RELEASE);
+ ::CloseHandle(section);
+ return;
+ }
+
+ // Map the file into the first placeholder.
+ PVOID file_view = ::MapViewOfFile3(
+ section, ::GetCurrentProcess(), placeholder, 0, file_region,
+ MEM_REPLACE_PLACEHOLDER, PAGE_READONLY, NULL, 0);
+ ::CloseHandle(section);
+ if (file_view == NULL) {
+ ::VirtualFree(placeholder, 0, MEM_RELEASE);
+ ::VirtualFree(static_cast<char *>(placeholder) + file_region,
+ 0, MEM_RELEASE);
+ return;
+ }
+
+ // Commit zero pages in the second placeholder (the padding).
+ void *pad = static_cast<char *>(placeholder) + file_region;
+ PVOID padding_ptr = ::VirtualAlloc2(
+ ::GetCurrentProcess(), pad, padding_region,
+ MEM_REPLACE_PLACEHOLDER | MEM_COMMIT, PAGE_READONLY, NULL, 0);
+ if (padding_ptr == NULL) {
+ ::UnmapViewOfFile(file_view);
+ ::VirtualFree(pad, 0, MEM_RELEASE);
+ return;
+ }
+
+ data = static_cast<const char *>(file_view);
+ padding_view_ = padding_ptr;
+}
+
+simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
+ if (data == nullptr) { return; }
+ ::UnmapViewOfFile(data);
+ if (padding_view_ != nullptr) {
+ ::VirtualFree(padding_view_, 0, MEM_RELEASE);
+ }
+}
+#endif // POSIX or _WIN32
simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noexcept simdjson_lifetime_bound {
if(!is_valid()) {
@@ -5112,7 +5757,8 @@ simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noe
simdjson_inline bool padded_memory_map::is_valid() const noexcept {
return data != nullptr;
}
-#endif // _WIN32
+
+#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
} // namespace simdjson
@@ -5222,6 +5868,9 @@ public:
simdjson_inline tape_ref() noexcept;
simdjson_inline tape_ref(const dom::document *doc, size_t json_index) noexcept;
inline size_t after_element() const noexcept;
+ // The reference must point to an element boundary inside the array whose
+ // opening tag is at array_start, or to that array's closing tag.
+ inline size_t before_element(size_t array_start) const noexcept;
simdjson_inline tape_type tape_ref_type() const noexcept;
simdjson_inline uint64_t tape_value() const noexcept;
simdjson_inline bool is_double() const noexcept;
@@ -5310,6 +5959,48 @@ public:
friend class array;
};
+ /**
+ * A forward iterator that visits the array's elements in reverse order.
+ * Like iterator, it returns element handles by value and does not own the
+ * document. No allocation or modification of the document is performed.
+ */
+ class reverse_iterator {
+ public:
+ using value_type = element;
+ using difference_type = std::ptrdiff_t;
+ using pointer = void;
+ using reference = value_type;
+ using iterator_category = std::forward_iterator_tag;
+
+ inline reference operator*() const noexcept;
+ inline reverse_iterator& operator++() noexcept;
+ inline reverse_iterator operator++(int) noexcept;
+ inline bool operator==(const reverse_iterator& other) const noexcept;
+ inline bool operator!=(const reverse_iterator& other) const noexcept;
+
+ reverse_iterator() noexcept = default;
+ reverse_iterator(const reverse_iterator&) noexcept = default;
+ reverse_iterator& operator=(const reverse_iterator&) noexcept = default;
+ private:
+ simdjson_inline reverse_iterator(const internal::tape_ref &tape, size_t array_start) noexcept;
+ internal::tape_ref tape{};
+ size_t array_start{};
+ friend class array;
+ };
+
+ /**
+ * Return the last array element, or rend() for an empty array.
+ * Incrementing the returned iterator moves toward the first element.
+ * A complete traversal takes O(n) time and O(1) additional space, where n
+ * is the number of immediate elements in the array.
+ * Finding the last or previous element can take O(n) time in the worst case when
+ * numeric payloads equal numeric type markers. Increments are amortized O(1)
+ * over a complete traversal; nested values do not increase this bound.
+ */
+ inline reverse_iterator rbegin() const noexcept;
+ /** Return the reverse traversal sentinel, before the first element. */
+ inline reverse_iterator rend() const noexcept;
+
/**
* Return the first array element.
*
@@ -5445,6 +6136,8 @@ public:
#if SIMDJSON_EXCEPTIONS
inline dom::array::iterator begin() const noexcept(false);
inline dom::array::iterator end() const noexcept(false);
+ inline dom::array::reverse_iterator rbegin() const noexcept(false);
+ inline dom::array::reverse_iterator rend() const noexcept(false);
inline size_t size() const noexcept(false);
#endif // SIMDJSON_EXCEPTIONS
};
@@ -5820,6 +6513,64 @@ public:
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_inline simdjson_result<element> parse(const char *buf) noexcept = delete;
+ /**
+ * Parse a JSON document whose buffer is **not** padded, in place and without
+ * copying it.
+ *
+ * *This feature is currently experimental.*
+ *
+ * The standard parse() methods require the input buffer to have at least
+ * SIMDJSON_PADDING extra readable bytes after the document (or they copy it
+ * into a padded buffer when realloc_if_needed is true). parse_unpadded() lifts
+ * that requirement: it parses directly from your buffer of exactly `len` bytes,
+ * never reading past `buf + len`, and never allocating a full padded copy.
+ *
+ * dom::parser parser;
+ * std::string_view json = get_json(); // no trailing padding needed
+ * dom::element doc = parser.parse_unpadded(json);
+ *
+ * This is the convenient way to use simdjson when you cannot (or do not want
+ * to) pad your input, e.g. a std::string_view into a larger buffer or a memory
+ * mapped file whose tail you do not control. It is generally a little slower
+ * than parsing a padded buffer with parse() (the very end of the document is
+ * handled with extra care), but it avoids the O(n) copy that
+ * parse(buf, len, true) performs when realloc_if_needed is true.
+ *
+ * The input is read but not modified, and it must remain valid (and the parser
+ * alive) for as long as you navigate the returned document, exactly like
+ * parse(buf, len, false).
+ *
+ * @param buf The JSON to parse. Only `len` bytes are read; no padding required.
+ * @param len The length of the JSON.
+ * @return An element pointing at the root of the document, or an error:
+ * - MEMALLOC if the parser does not have enough capacity and allocation fails.
+ * - CAPACITY if the parser does not have enough capacity and len > max_capacity.
+ * - other json errors if parsing fails.
+ */
+ inline simdjson_result<element> parse_unpadded(const uint8_t *buf, size_t len) & noexcept;
+ inline simdjson_result<element> parse_unpadded(const uint8_t *buf, size_t len) && =delete;
+ /** @overload parse_unpadded(const uint8_t *buf, size_t len) */
+ simdjson_inline simdjson_result<element> parse_unpadded(const char *buf, size_t len) & noexcept;
+ simdjson_inline simdjson_result<element> parse_unpadded(const char *buf, size_t len) && =delete;
+ /** @overload parse_unpadded(const uint8_t *buf, size_t len) */
+ simdjson_inline simdjson_result<element> parse_unpadded(std::string_view s) & noexcept;
+ simdjson_inline simdjson_result<element> parse_unpadded(std::string_view s) && =delete;
+
+ /**
+ * Parse a non-padded JSON document into a caller-provided document instance, in
+ * place and without copying. This is to parse_unpadded() what
+ * parse_into_document() is to parse(). See parse_unpadded() for the padding and
+ * lifetime semantics.
+ *
+ * *This feature is currently experimental.*
+ *
+ * @param doc The document instance where the parsed data will be stored (on success).
+ * @param buf The JSON to parse. Only `len` bytes are read; no padding required.
+ * @param len The length of the JSON.
+ */
+ inline simdjson_result<element> parse_into_document_unpadded(document& doc, const uint8_t *buf, size_t len) & noexcept;
+ inline simdjson_result<element> parse_into_document_unpadded(document& doc, const uint8_t *buf, size_t len) && =delete;
+
/**
* Parse a JSON document into a provide document instance and return a temporary reference to it.
* It is similar to the function `parse` except that instead of parsing into the internal
@@ -6045,7 +6796,7 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
@@ -6057,14 +6808,50 @@ public:
inline simdjson_result<document_stream> parse_many(const char *buf, size_t len, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
inline simdjson_result<document_stream> parse_many(const std::string &s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
- inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size) = delete;// unsafe
+ inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
- inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size) = delete;// unsafe
+ inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ *
+ * Because padded_string_view guarantees SIMDJSON_PADDING trailing bytes, this
+ * overload is safe to use with buffers that the caller owns elsewhere (for
+ * example, a padded_memory_map), with no extra copy. Without this overload,
+ * passing a padded_string_view would silently bind to the padded_string
+ * overload via an implicit conversion, allocating and copying the input, and
+ * -- because that temporary is destroyed at the end of the full-expression --
+ * leaving the returned document_stream pointing at freed memory. */
+ inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> parse_many(const char *buf, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept = delete;
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format.
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> parse_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> parse_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while
+ * the returned document_stream only holds a pointer to it: iterating the stream would
+ * then read freed memory. These deleted overloads also catch a std::string_view
+ * argument, which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload parse_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/**
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
* and `max_depth` depth.
@@ -6354,8 +7141,9 @@ public:
*
* IMPORTANT: this value is only meaningful under the conditions below. It is
* computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * merely imprecise, it is
+ * arbitrary -- it can exceed size_in_bytes() or wrap around to a huge value.
+ * Check it only when all of the following hold:
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -6364,6 +7152,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
/**
@@ -6463,12 +7254,14 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param format is the stream format
*/
simdjson_inline document_stream(
dom::parser &parser,
const uint8_t *buf,
size_t len,
- size_t batch_size
+ size_t batch_size,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -6518,6 +7311,8 @@ private:
const uint8_t *buf;
size_t len;
size_t batch_size;
+ /** The stream format. */
+ stream_format format;
/** The error (or lack thereof) from the current document. */
error_code error;
size_t batch_start{0};
@@ -6605,6 +7400,8 @@ enum class element_type {
STRING = '"', ///< std::string_view
BOOL = 't', ///< bool
NULL_VALUE = 'n', ///< null
+ /// The BIGINT type is for integers that do not fit in 64 bits. It is only present
+ // if you set parser.number_as_string(true).
BIGINT = 'Z' ///< std::string_view: big integer stored as raw digit string
};
@@ -6673,6 +7470,20 @@ public:
* Returns INCORRECT_TYPE if the JSON element is not a string.
*/
inline simdjson_result<std::string_view> get_string() const noexcept;
+
+ #if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this element to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * @returns A std::u8string_view. The string is stored in the parser and will be invalidated the next time it
+ * parses a document or when it is destroyed.
+ * Returns INCORRECT_TYPE if the JSON element is not a string.
+ */
+ inline simdjson_result<std::u8string_view> get_u8string() const noexcept;
+ #endif
+
/**
* Cast this element to a signed integer.
*
@@ -6778,7 +7589,7 @@ public:
* Supported types:
* - Boolean: bool
* - Number: double, uint64_t, int64_t
- * - String: std::string_view, const char *
+ * - String: std::string_view, const char *, std::u8string_view (C++20)
* - Array: dom::array
* - Object: dom::object
*
@@ -6793,7 +7604,7 @@ public:
* Supported types:
* - Boolean: bool
* - Number: double, uint64_t, int64_t
- * - String: std::string_view, const char *
+ * - String: std::string_view, const char *, std::u8string_view (C++20)
* - Array: dom::array
* - Object: dom::object
*
@@ -6823,7 +7634,7 @@ public:
* Supported types:
* - Boolean: bool
* - Number: double, uint64_t, int64_t
- * - String: std::string_view, const char *
+ * - String: std::string_view, const char *, std::u8string_view (C++20)
* - Array: dom::array
* - Object: dom::object
*
@@ -6842,7 +7653,7 @@ public:
* Supported types:
* - Boolean: bool
* - Number: double, uint64_t, int64_t
- * - String: std::string_view, const char *
+ * - String: std::string_view, const char *, std::u8string_view (C++20)
* - Array: dom::array
* - Object: dom::object
*
@@ -7126,6 +7937,9 @@ public:
simdjson_inline simdjson_result<const char *> get_c_str() const noexcept;
simdjson_inline simdjson_result<size_t> get_string_length() const noexcept;
simdjson_inline simdjson_result<std::string_view> get_string() const noexcept;
+ #if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string() const noexcept;
+ #endif
simdjson_inline simdjson_result<int64_t> get_int64() const noexcept;
simdjson_inline simdjson_result<uint64_t> get_uint64() const noexcept;
simdjson_inline simdjson_result<double> get_double() const noexcept;
@@ -7825,6 +8639,24 @@ template <class T> std::string prettify(simdjson_result<T> x) {
namespace simdjson {
+/** Specifies where commas should be in table-formatted elements. */
+enum class table_comma_placement {
+ /** Commas come right after the value */
+ before_padding,
+ /** Commas come after the column padding, so they line up in their own column. */
+ after_padding,
+ /** Commas come right after the value, except for columns of numbers */
+ before_padding_except_numbers,
+};
+
+/** Options for how lists or columns of numbers should be aligned */
+enum class number_list_alignment {
+ /** Left-aligns numbers */
+ left,
+ /** Right-aligns numbers */
+ right,
+};
+
/**
* Configuration options for FracturedJson formatting.
*
@@ -7839,12 +8671,6 @@ struct fractured_json_options {
*/
size_t max_total_line_length = 120;
- /**
- * Maximum length for inlined elements (default: 80).
- * Simple arrays/objects shorter than this may be rendered inline.
- */
- size_t max_inline_length = 80;
-
/**
* Maximum nesting depth for inline rendering (default: 2).
* Elements with complexity exceeding this will be expanded.
@@ -7853,11 +8679,11 @@ struct fractured_json_options {
size_t max_inline_complexity = 2;
/**
- * Maximum complexity for compact array formatting (default: 1).
+ * Maximum complexity for compact array formatting (default: 2).
* Arrays with elements of this complexity or less may have multiple
* items per line.
*/
- size_t max_compact_array_complexity = 1;
+ size_t max_compact_array_complexity = 2;
/**
* Number of spaces per indentation level (default: 4).
@@ -7865,23 +8691,26 @@ struct fractured_json_options {
size_t indent_spaces = 4;
/**
- * Enable tabular formatting for arrays of similar objects (default: true).
- * When enabled, arrays of objects with identical keys are formatted
- * as aligned tables.
+ * Forces elements close to the root to always fully expand, regardless of other settings.
+ * (default: -1). -1 = none; 0 = root node only; 1 = root node and its children; etc.
*/
- bool enable_table_format = true;
+ int always_expand_depth = -1;
/**
- * Minimum number of rows to trigger table mode (default: 3).
+ * Enable tabular formatting for arrays of similar objects or arrays
+ * (default: true). When enabled, the rows of such an array are written one
+ * per line with their columns aligned. Rows need not have identical keys:
+ * columns are ordered by the first occurrence of each key, and a row
+ * missing a key gets blank space in that column.
*/
- size_t min_table_rows = 3;
+ bool enable_table_format = true;
/**
- * Similarity threshold for table detection (default: 0.8).
- * Objects must share at least this fraction of keys to be formatted
- * as a table.
+ * Maximum complexity of each row of a table (default: 2).
+ * 0 = rows may only be scalars (a single column); 1 = rows may be flat
+ * arrays/objects; higher values allow deeper nesting.
*/
- double table_similarity_threshold = 0.8;
+ size_t max_table_row_complexity = 2;
/**
* Enable compact multiline arrays (default: true).
@@ -7891,16 +8720,26 @@ struct fractured_json_options {
bool enable_compact_multiline = true;
/**
- * Maximum array items per line in compact mode (default: 10).
+ * Minimum number of items per line for an array to be formatted as a
+ * compact multiline array (default: 3).
*/
- size_t max_items_per_line = 10;
+ size_t min_compact_array_row_items = 3;
/**
- * Add space inside brackets for simple containers (default: true).
- * When true: { "key": "value" }
- * When false: {"key": "value"}
+ * Add space inside brackets for containers that hold only scalar values
+ * (default: false). When true: { "key": "value" }. When false:
+ * {"key": "value"}.
+ * @see nested_bracket_padding
*/
- bool simple_bracket_padding = true;
+ bool simple_bracket_padding = false;
+
+ /**
+ * Add space inside brackets for containers that hold at least one
+ * nested array/object (default: true). When true: { "a": [1, 2] }.
+ * When false: {"a": [1, 2]}.
+ * @see simple_bracket_padding
+ */
+ bool nested_bracket_padding = true;
/**
* Add space after colons (default: true).
@@ -7915,6 +8754,18 @@ struct fractured_json_options {
* When false: [1,2,3]
*/
bool comma_padding = true;
+
+ /**
+ * Placement of commas relative to column padding in table-formatted rows
+ * and compact multiline arrays (default: before_padding_except_numbers).
+ */
+ table_comma_placement comma_placement = table_comma_placement::before_padding_except_numbers;
+
+ /**
+ * Controls alignment of numbers in table columns or compact multiline arrays
+ * (default: left). Numbers are always written exactly as in the input.
+ */
+ number_list_alignment number_alignment = number_list_alignment::left;
};
/**
@@ -7995,12 +8846,55 @@ inline std::string fractured_json_string(std::string_view json_str,
#ifndef SIMDJSON_JSONPATHUTIL_H
#define SIMDJSON_JSONPATHUTIL_H
+/* skipped duplicate #include "simdjson/error.h" */
#include <string>
/* skipped duplicate #include "simdjson/common_defs.h" */
+#include <limits>
#include <utility>
namespace simdjson {
+namespace internal {
+/**
+ * Parses the next JSON Pointer array index token.
+ *
+ * The caller passes a pointer fragment with no leading '/', such as "123/foo".
+ * On success, array_index receives the parsed index and token_length receives
+ * the number of bytes consumed before the next '/' or the end of the fragment.
+ */
+simdjson_inline error_code parse_json_pointer_array_index(std::string_view json_pointer,
+ size_t &array_index,
+ size_t &token_length) noexcept {
+ array_index = 0;
+ token_length = 0;
+
+ for (; token_length < json_pointer.length() && json_pointer[token_length] != '/';
+ token_length++) {
+ uint8_t digit = uint8_t(json_pointer[token_length] - '0');
+ // Check for non-digit in array index. If it's there, we're trying to get a field in an object.
+ if (digit > 9) {
+ return INCORRECT_TYPE;
+ }
+ // 0 followed by other digits is invalid.
+ if (token_length > 0 && json_pointer[0] == '0') {
+ return INVALID_JSON_POINTER;
+ }
+ if (array_index >
+ (((std::numeric_limits<size_t>::max)() - digit) / 10)) {
+ return INDEX_OUT_OF_BOUNDS;
+ }
+ array_index = array_index * 10 + digit;
+ }
+
+ // Empty string is invalid; so is a "/" with no digits before it.
+ if (token_length == 0) {
+ return INVALID_JSON_POINTER;
+ }
+
+ return SUCCESS;
+}
+} // namespace internal
+
/**
* Converts JSONPath to JSON Pointer.
* @param json_path The JSONPath string to be converted.
@@ -8220,6 +9114,72 @@ inline size_t tape_ref::after_element() const noexcept {
simdjson_inline tape_type tape_ref::tape_ref_type() const noexcept {
return static_cast<tape_type>(doc->tape[json_index] >> 56);
}
+simdjson_inline size_t tape_ref::before_element(size_t array_start) const noexcept {
+ SIMDJSON_DEVELOPMENT_ASSERT(usable());
+ SIMDJSON_DEVELOPMENT_ASSERT(json_index > array_start);
+ tape_ref previous(doc, json_index - 1);
+ if (previous.json_index == array_start) { return array_start; }
+ // An exact numeric marker cannot end an element unless it is itself the
+ // payload of a number. In that case its header is immediately before it.
+ if (previous.is_int64() || previous.is_uint64() || previous.is_double()) {
+ return previous.json_index - 1;
+ }
+ tape_ref probe(doc, previous.json_index - 1);
+
+ // Validate both container links before examining its contents. A candidate
+ // opening tag preceded by an even run of numeric markers is a real tag,
+ // not a numeric payload. Scan both candidate boundaries together so that a
+ // forged opening tag inside a nested value cannot cause an unbounded detour.
+ // For a real container, the opening probe visits preceding siblings. For a
+ // numeric payload, the other probe does. Stopping at the shorter run bounds
+ // the work by the array's immediate elements, rather than nested contents.
+ const auto type = previous.tape_ref_type();
+ if (type == tape_type::END_ARRAY || type == tape_type::END_OBJECT) {
+ const size_t start = previous.matching_brace_index();
+ if (start > array_start && start < previous.json_index) {
+ tape_ref opening(doc, start);
+ const auto expected = type == tape_type::END_ARRAY
+ ? tape_type::START_ARRAY : tape_type::START_OBJECT;
+ if (opening.tape_ref_type() == expected &&
+ opening.matching_brace_index() == previous.json_index + 1) {
+ tape_ref before_opening(doc, start - 1);
+ while ((before_opening.is_int64() || before_opening.is_uint64() ||
+ before_opening.is_double()) &&
+ (probe.is_int64() || probe.is_uint64() || probe.is_double())) {
+ --before_opening.json_index;
+ --probe.json_index;
+ }
+ if (!before_opening.is_int64() && !before_opening.is_uint64() &&
+ !before_opening.is_double() &&
+ (start - before_opening.json_index) % 2 == 1) {
+ return start;
+ }
+ }
+ }
+ }
+
+ // Numeric payloads can have ANY bit pattern, including another type's tag.
+ // A run of exact numeric markers starts with a header, then alternates
+ // between payload and header. An odd run before this word makes it a payload.
+ // Subsequent reverse increments through exact-marker payloads take the
+ // constant-time numeric-payload branch above.
+ while (probe.is_int64() || probe.is_uint64() || probe.is_double()) {
+ --probe.json_index;
+ }
+ if ((previous.json_index - probe.json_index) % 2 == 0) {
+ return previous.json_index - 1;
+ }
+
+ // Once distinguished from numeric payloads, closing container tags link
+ // directly back to their opening tags.
+ switch (previous.tape_ref_type()) {
+ case tape_type::END_ARRAY:
+ case tape_type::END_OBJECT:
+ return previous.matching_brace_index();
+ default:
+ return previous.json_index;
+ }
+}
simdjson_inline uint64_t internal::tape_ref::tape_value() const noexcept {
return doc->tape[json_index] & internal::JSON_VALUE_MASK;
}
@@ -8295,6 +9255,14 @@ inline size_t simdjson_result<dom::array>::size() const noexcept(false) {
if (error()) { throw simdjson_error(error()); }
return first.size();
}
+inline dom::array::reverse_iterator simdjson_result<dom::array>::rbegin() const noexcept(false) {
+ if (error()) { throw simdjson_error(error()); }
+ return first.rbegin();
+}
+inline dom::array::reverse_iterator simdjson_result<dom::array>::rend() const noexcept(false) {
+ if (error()) { throw simdjson_error(error()); }
+ return first.rend();
+}
#endif // SIMDJSON_EXCEPTIONS
@@ -8304,6 +9272,7 @@ inline simdjson_result<dom::element> simdjson_result<dom::array>::at_pointer(std
}
inline simdjson_result<dom::element> simdjson_result<dom::array>::at_path(std::string_view json_path) const noexcept {
+ if (error()) { return error(); }
auto json_pointer = json_path_to_pointer_conversion(json_path);
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
return at_pointer(json_pointer);
@@ -8344,6 +9313,15 @@ inline size_t array::size() const noexcept {
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
return tape.scope_count();
}
+inline array::reverse_iterator array::rbegin() const noexcept {
+ SIMDJSON_DEVELOPMENT_ASSERT(tape.usable());
+ const internal::tape_ref end_tape(tape.doc, tape.matching_brace_index() - 1);
+ return reverse_iterator(internal::tape_ref(tape.doc, end_tape.before_element(tape.json_index)), tape.json_index);
+}
+inline array::reverse_iterator array::rend() const noexcept {
+ SIMDJSON_DEVELOPMENT_ASSERT(tape.usable());
+ return reverse_iterator(tape, tape.json_index);
+}
inline size_t array::number_of_slots() const noexcept {
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
return tape.matching_brace_index() - tape.json_index;
@@ -8360,21 +9338,9 @@ inline simdjson_result<element> array::at_pointer(std::string_view json_pointer)
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = array(tape).at(array_index);
@@ -8409,7 +9375,6 @@ inline void array::process_json_path_of_child_elements(std::vector<element>::ite
if(error) {
continue;
}
- accumulator.reserve(accumulator.size() + child_result.size());
accumulator.insert(accumulator.end(),
std::make_move_iterator(child_result.begin()),
std::make_move_iterator(child_result.end()));
@@ -8505,6 +9470,30 @@ inline array::operator element() const noexcept {
return element(tape);
}
+//
+// array::reverse_iterator inline implementation
+//
+simdjson_inline array::reverse_iterator::reverse_iterator(const internal::tape_ref &_tape, size_t _array_start) noexcept
+ : tape{_tape}, array_start{_array_start} { }
+inline element array::reverse_iterator::operator*() const noexcept {
+ return element(tape);
+}
+inline array::reverse_iterator& array::reverse_iterator::operator++() noexcept {
+ tape.json_index = tape.before_element(array_start);
+ return *this;
+}
+inline array::reverse_iterator array::reverse_iterator::operator++(int) noexcept {
+ reverse_iterator out = *this;
+ ++*this;
+ return out;
+}
+inline bool array::reverse_iterator::operator==(const reverse_iterator& other) const noexcept {
+ return tape.doc == other.tape.doc && tape.json_index == other.tape.json_index;
+}
+inline bool array::reverse_iterator::operator!=(const reverse_iterator& other) const noexcept {
+ return !(*this == other);
+}
+
//
// array::iterator inline implementation
//
@@ -8596,6 +9585,7 @@ inline simdjson_result<dom::element> simdjson_result<dom::object>::at_pointer(st
return first.at_pointer(json_pointer);
}
inline simdjson_result<dom::element> simdjson_result<dom::object>::at_path(std::string_view json_path) const noexcept {
+ if (error()) { return error(); }
auto json_pointer = json_path_to_pointer_conversion(json_path);
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
return at_pointer(json_pointer);
@@ -8725,7 +9715,6 @@ inline void object::process_json_path_of_child_elements(std::vector<element>::it
if(error) {
continue;
}
- accumulator.reserve(accumulator.size() + child_result.size());
accumulator.insert(accumulator.end(),
std::make_move_iterator(child_result.begin()),
std::make_move_iterator(child_result.end()));
@@ -9003,6 +9992,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<dom::element>:
if (error()) { return error(); }
return first.get_string();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<dom::element>::get_u8string() const noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string();
+}
+#endif
simdjson_inline simdjson_result<int64_t> simdjson_result<dom::element>::get_int64() const noexcept {
if (error()) { return error(); }
return first.get_int64();
@@ -9069,6 +10064,7 @@ simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_
return first.at_pointer(json_pointer);
}
simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_path(const std::string_view json_path) const noexcept {
+ if (error()) { return error(); }
auto json_pointer = json_path_to_pointer_conversion(json_path);
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
return at_pointer(json_pointer);
@@ -9201,6 +10197,13 @@ inline simdjson_result<std::string_view> element::get_string() const noexcept {
return INCORRECT_TYPE;
}
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+inline simdjson_result<std::u8string_view> element::get_u8string() const noexcept {
+ std::string_view v;
+ SIMDJSON_TRY(get_string().get(v));
+ return internal::as_u8string_view(v);
+}
+#endif
inline simdjson_result<uint64_t> element::get_uint64() const noexcept {
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
if(simdjson_unlikely(!tape.is_uint64())) { // branch rarely taken
@@ -9296,6 +10299,9 @@ template<> inline simdjson_result<array> element::get<array>() const noexcept {
template<> inline simdjson_result<object> element::get<object>() const noexcept { return get_object(); }
template<> inline simdjson_result<const char *> element::get<const char *>() const noexcept { return get_c_str(); }
template<> inline simdjson_result<std::string_view> element::get<std::string_view>() const noexcept { return get_string(); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> inline simdjson_result<std::u8string_view> element::get<std::u8string_view>() const noexcept { return get_u8string(); }
+#endif
template<> inline simdjson_result<int64_t> element::get<int64_t>() const noexcept { return get_int64(); }
template<> inline simdjson_result<uint64_t> element::get<uint64_t>() const noexcept { return get_uint64(); }
template<> inline simdjson_result<double> element::get<double>() const noexcept { return get_double(); }
@@ -9628,6 +10634,29 @@ inline simdjson_result<element> parser::parse_into_document(document& provided_d
return provided_doc.root();
}
+inline simdjson_result<element> parser::parse_into_document_unpadded(document& provided_doc, const uint8_t *buf, size_t len) & noexcept {
+ // Like parse_into_document with realloc_if_needed=false (no copy, parse in
+ // place), but we tell the implementation the buffer is not padded so stage 2
+ // avoids reading past buf+len: string unescaping is bounded, near-the-end
+ // numbers are parsed from a padded copy, and atoms use length-aware
+ // validators (see tape_builder). Stage 1 is already safe for unpadded input.
+ error_code _error = ensure_capacity(provided_doc, len);
+ if (_error) { return _error; }
+
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ implementation->_number_as_string = _number_as_string;
+ implementation->_unpadded = true;
+ _error = implementation->parse(buf, len, provided_doc);
+ implementation->_unpadded = false; // restore so later padded parses use the fast path
+
+ if (_error) { return _error; }
+
+ return provided_doc.root();
+}
+
simdjson_inline simdjson_result<element> parser::parse_into_document(document& provided_doc, const char *buf, size_t len, bool realloc_if_needed) & noexcept {
return parse_into_document(provided_doc, reinterpret_cast<const uint8_t *>(buf), len, realloc_if_needed);
}
@@ -9656,13 +10685,18 @@ simdjson_inline simdjson_result<element> parser::parse(const padded_string_view
return parse(v.data(), v.length(), false);
}
+inline simdjson_result<element> parser::parse_unpadded(const uint8_t *buf, size_t len) & noexcept {
+ return parse_into_document_unpadded(doc, buf, len);
+}
+simdjson_inline simdjson_result<element> parser::parse_unpadded(const char *buf, size_t len) & noexcept {
+ return parse_unpadded(reinterpret_cast<const uint8_t *>(buf), len);
+}
+simdjson_inline simdjson_result<element> parser::parse_unpadded(std::string_view s) & noexcept {
+ return parse_unpadded(reinterpret_cast<const uint8_t *>(s.data()), s.size());
+}
+
inline simdjson_result<document_stream> parser::parse_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
- if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
- if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
- buf += 3;
- len -= 3;
- }
- return document_stream(*this, buf, len, batch_size);
+ return parse_many(buf, len, batch_size, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::parse_many(const char *buf, size_t len, size_t batch_size) noexcept {
return parse_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
@@ -9673,6 +10707,48 @@ inline simdjson_result<document_stream> parser::parse_many(const std::string &s,
inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size) noexcept {
return parse_many(s.data(), s.length(), batch_size);
}
+inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size) noexcept {
+ return parse_many(v.data(), v.length(), batch_size);
+}
+
+inline simdjson_result<document_stream> parser::parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return parse_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return parse_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return parse_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept {
+ return parse_many(v.data(), v.length(), batch_size, format);
+}
simdjson_inline size_t parser::capacity() const noexcept {
return implementation ? implementation->capacity() : 0;
@@ -9830,12 +10906,14 @@ simdjson_inline document_stream::document_stream(
dom::parser &_parser,
const uint8_t *_buf,
size_t _len,
- size_t _batch_size
+ size_t _batch_size,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
@@ -9853,6 +10931,7 @@ simdjson_inline document_stream::document_stream() noexcept
buf{nullptr},
len{0},
batch_size{0},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -9929,6 +11008,7 @@ inline void document_stream::start() noexcept {
if (error) { return; }
error = parser->ensure_capacity(batch_size);
if (error) { return; }
+ parser->implementation->_number_as_string = parser->number_as_string();
// Always run the first stage 1 parse immediately
batch_start = 0;
error = run_stage1(*parser, batch_start);
@@ -9968,7 +11048,40 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
} else {
size_t next_doc_index = stream->batch_start + stream->parser->implementation->structural_indexes[stream->parser->implementation->next_structural_index];
size_t svlen = next_doc_index - current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_doc_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464 json_sequence
+ // mode the scanner classifies RS as a scalar character, so an RS-prefixed
+ // scalar document (number/true/false/null/string) has no closing structural
+ // index and the slice runs all the way up to the next document's RS. RS
+ // cannot legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping it is
+ // safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -10008,6 +11121,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -10018,10 +11134,35 @@ inline size_t document_stream::next_batch_start() const noexcept {
inline error_code document_stream::run_stage1(dom::parser &p, size_t _batch_start) noexcept {
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -10194,16 +11335,25 @@ inline error_code document::allocate(size_t capacity) noexcept {
allocated_capacity = 0;
return SUCCESS;
}
+ if (capacity > SIMDJSON_MAXSIZE_BYTES) {
+ return CAPACITY;
+ }
// a pathological input like "[[[[..." would generate capacity tape elements, so
// need a capacity of at least capacity + 1, but it is also possible to do
// worse with "[7,7,7,7,6,7,7,7,6,7,7,6,[7,7,7,7,6,7,7,7,6,7,7,6,7,7,7,7,7,7,6"
//where capacity + 1 tape elements are
// generated, see issue https://github.com/simdjson/simdjson/issues/345
+ if(capacity + 3 < capacity) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
size_t tape_capacity = SIMDJSON_ROUNDUP_N(capacity + 3, 64);
// a document with only zero-length strings... could have capacity/3 string
// and we would need capacity/3 * 5 bytes on the string buffer
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset( new (std::nothrow) uint8_t[string_capacity]);
tape.reset(new (std::nothrow) uint64_t[tape_capacity]);
if(!(string_buf && tape)) {
@@ -10347,6 +11497,7 @@ inline bool document::dump_raw_tape(std::ostream &os) const noexcept {
/* skipped duplicate #include "simdjson/dom/object-inl.h" */
/* skipped duplicate #include "simdjson/internal/tape_ref-inl.h" */
+#include <cmath>
#include <cstring>
namespace simdjson {
@@ -10517,11 +11668,29 @@ simdjson_inline void base_formatter<formatter>::number(int64_t x) {
template <class formatter>
simdjson_inline void base_formatter<formatter>::number(double x) {
- char number_buffer[24];
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(x))) {
+ if (std::isnan(x)) {
+ char const *s = "NaN";
+ chars(s, s + 3);
+ } else {
+ if (x < 0) {
+ one_char('-');
+ }
+ char const *s = "Infinity";
+ chars(s, s + 8);
+ }
+ return;
+ }
+#endif
+
+ // Must be to_chars_buffer_size (40): only ~24 chars are emitted, but
+ // to_chars over-writes with fixed-size 16/17-byte copies for inlining.
+ char number_buffer[simdjson::internal::to_chars_buffer_size];
// Currently, passing the nullptr to the second argument is
// safe because our implementation does not check the second
// argument.
- char *newp = internal::to_chars(number_buffer, nullptr, x);
+ char *newp = simdjson::internal::to_chars(number_buffer, nullptr, x);
chars(number_buffer, newp);
}
@@ -10797,7 +11966,7 @@ inline void string_builder<serializer>::append(simdjson::dom::element value) {
format.string(iter.get_string_view());
break;
case tape_type::BIGINT: {
- // Big integer stored as string — output raw digits (no quotes)
+ // Big integer stored as string -- output raw digits (no quotes)
auto sv = iter.get_string_view();
format.chars(sv.data(), sv.data() + sv.size());
break;
@@ -10935,7 +12104,7 @@ simdjson_inline std::string_view string_builder<serializer>::str() const {
#include <vector>
#include <string>
#include <string_view>
-#include <set>
+#include <utility>
namespace simdjson {
namespace internal {
@@ -10944,12 +12113,50 @@ namespace internal {
* Layout mode for fractured JSON formatting.
*/
enum class layout_mode {
- INLINE, // Single line: [1, 2, 3] or {"a": 1}
- COMPACT_MULTILINE, // Multiple items per line with breaks
- TABLE, // Tabular format for arrays of similar objects
- EXPANDED // Traditional multi-line with indentation
+ single_line, // Single line: [1, 2, 3] or {"a": 1}
+ compact_multiline, // Multiple items per line with breaks
+ table, // Tabular format for arrays of similar objects
+ expanded // Traditional multi-line with indentation
};
+/** Kind of value found in a table column across all rows that have one. */
+enum class table_column_type {
+ unknown,
+ simple, // string, bool, or null
+ number,
+ array,
+ object,
+ mixed // rows disagree on kind
+};
+
+/** Column of a table-formatted array.*/
+struct table_column {
+ /** Column name for object rows (may be the empty string). */
+ std::string key{};
+ /** True for object rows (the column has a key), false for array rows. */
+ bool has_key = false;
+ /** Rendered length of key */
+ size_t key_width = 0;
+ table_column_type type = table_column_type::unknown;
+ /** Widest rendered value in this column (when fully expanding all children) */
+ size_t width = 0;
+ /** Widest plain value in this column. */
+ size_t plain_width = 0;
+ /** subcolumns (only populated if every child is an array or every child is an object) */
+ std::vector<table_column> children{};
+};
+
+/** Whether rows from this column list use nested vs. simple bracket padding
+ * One shared decision, since picking it per-row would misalign width-aligned rows. */
+inline bool table_row_is_nested(const std::vector<table_column>& columns) {
+ for (const table_column& col : columns) {
+ if (col.type == table_column_type::object || col.type == table_column_type::array) {
+ return true;
+ }
+ }
+ return false;
+}
+
/**
* Metrics computed for a JSON element during structure analysis.
* These metrics drive layout decisions and contain child metrics for recursive formatting.
@@ -10970,16 +12177,32 @@ struct element_metrics {
/** Is this an array where all elements have similar structure? */
bool is_uniform_array = false;
- /** For uniform arrays of objects: the common keys */
- std::vector<std::string> common_keys{};
-
- /** Recommended layout mode based on analysis */
- layout_mode recommended_layout = layout_mode::EXPANDED;
+ /** for uniform arrays: this array's columns for alignment */
+ std::vector<table_column> table_columns{};
+ /** Widest table row after pruning recursive columns that don't fit into line budget */
+ size_t table_row_width = 0;
+ /** Widest table row without pruning recursive columns that don't fit into line budget */
+ size_t table_row_width_full = 0;
/** Child metrics for arrays and objects (in order of iteration) */
std::vector<element_metrics> children{};
+
+ /** For scalar uniform arrays (table_columns empty): the rows' common type. */
+ table_column_type scalar_column_type = table_column_type::unknown;
};
+/** children[idx], or a default-constructed element_metrics if idx is out of range */
+inline const element_metrics& child_metrics_at(const std::vector<element_metrics>& children, size_t idx) {
+ static const element_metrics empty{};
+ return idx < children.size() ? children[idx] : empty;
+}
+
+/** *ptr, or a default-constructed element_metrics if ptr is null. */
+inline const element_metrics& child_metrics_at(const element_metrics* ptr) {
+ static const element_metrics empty{};
+ return ptr ? *ptr : empty;
+}
+
/**
* Analyzes JSON structure to compute metrics for formatting decisions.
*
@@ -11040,20 +12263,27 @@ public:
element_metrics analyze_object(const dom::object& obj,
const fractured_json_options& opts);
+ /** Decide layout at the given render depth. Kept out of analysis since
+ * the same metrics can render inline or expanded at different depths. */
+ static layout_mode decide_layout(const element_metrics& metrics,
+ size_t depth,
+ const fractured_json_options& opts,
+ bool has_trailing_comma = false);
+
private:
const fractured_json_options* current_opts_ = nullptr;
/** Recursive analysis implementation */
- element_metrics analyze_element(const dom::element& elem, size_t depth);
+ element_metrics analyze_element(const dom::element& elem, size_t depth) const;
/** Analyze scalar values (strings, numbers, booleans, null) */
- element_metrics analyze_scalar(const dom::element& elem);
+ element_metrics analyze_scalar(const dom::element& elem) const;
/** Analyze an array element */
- element_metrics analyze_array(const dom::array& arr, size_t depth);
+ element_metrics analyze_array(const dom::array& arr, size_t depth) const;
/** Analyze an object element */
- element_metrics analyze_object(const dom::object& obj, size_t depth);
+ element_metrics analyze_object(const dom::object& obj, size_t depth) const;
/** Estimate inline length for a string (including quotes and escaping) */
size_t estimate_string_length(std::string_view s) const;
@@ -11064,27 +12294,35 @@ private:
size_t estimate_number_length(uint64_t u) const;
/**
- * Check if an array contains uniform objects suitable for table formatting.
+ * Check if an array contains uniformly-shaped rows suitable for table
+ * formatting, filling in corresponding metrics
* @param arr The array to check
- * @param common_keys Output: keys common to all objects
- * @return true if the array is suitable for table formatting
+ * @param metrics The array's metrics, with children already filled in
+ * @param depth The array's depth
*/
- bool check_array_uniformity(const dom::array& arr,
- std::vector<std::string>& common_keys) const;
+ bool check_array_uniformity(const dom::array& arr, element_metrics& metrics, size_t depth) const;
- /**
- * Compute similarity between two objects.
- * @return Fraction of keys that are common (0.0 to 1.0)
- */
- double compute_object_similarity(const dom::object& a,
- const dom::object& b) const;
+ static table_column_type classify_table_value(dom::element_type type);
- /**
- * Decide the recommended layout mode based on metrics and options.
- */
- layout_mode decide_layout(const element_metrics& metrics,
- size_t depth,
- size_t available_width) const;
+ /** Find common type across a set of sibling values and the widest of their rendered lengths */
+ static void classify_and_measure(const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+ table_column_type& common, size_t& max_width);
+
+ /** Recursively build table columns for an array */
+ void build_table_columns(const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+ std::vector<table_column>& out_columns) const;
+
+ /** Rendered width of a row, assuming each column's current */
+ size_t compute_columns_width(const std::vector<table_column>& columns) const;
+
+ /** Height of a column's recursion (0 = leaf). */
+ static size_t column_height(const table_column& column);
+
+ /** Flatten deepest columns of the table */
+ static bool flatten_deepest_columns(std::vector<table_column>& columns);
+
+ /** Bottom-up refresh of column widths after flatten_deepest_columns. */
+ void recompute_column_widths(std::vector<table_column>& columns) const;
};
} // namespace internal
@@ -11130,56 +12368,35 @@ public:
/** Get the current layout mode */
layout_mode get_layout_mode() const;
- /** Set current depth for formatting decisions */
- void set_depth(size_t depth);
-
- /** Get current depth */
- size_t get_depth() const;
-
/** Track current line length for compact multiline decisions */
void track_line_length(size_t chars);
- /** Reset line length (after newline) */
- void reset_line_length();
-
- /** Get current line length */
- size_t get_line_length() const;
-
/** Check if we should break to a new line in compact mode */
bool should_break_line(size_t upcoming_length) const;
/** Get the options */
const fractured_json_options& options() const;
- // Table formatting support
- /** Begin a table row */
- void begin_table_row();
-
- /** End a table row */
- void end_table_row();
-
- /** Set column widths for table alignment */
- void set_column_widths(const std::vector<size_t>& widths);
-
- /** Get current column index in table mode */
- size_t get_column_index() const;
-
- /** Advance to next column */
- void next_column();
-
- /** Add padding to align with column width */
- void align_to_column_width(size_t actual_width);
-
private:
fractured_json_options options_;
- layout_mode current_layout_ = layout_mode::EXPANDED;
- size_t current_depth_ = 0;
+ layout_mode current_layout_ = layout_mode::expanded;
size_t current_line_length_ = 0;
+};
- // Table state
- bool in_table_mode_ = false;
- std::vector<size_t> column_widths_;
- size_t current_column_ = 0;
+/** RAII helper forcing single line layout and restoring previous mode on exit */
+class scoped_single_line_mode {
+public:
+ explicit scoped_single_line_mode(fractured_formatter& format)
+ : format_(format), prev_(format.get_layout_mode()) {
+ format_.set_layout_mode(layout_mode::single_line);
+ }
+ ~scoped_single_line_mode() { format_.set_layout_mode(prev_); }
+ scoped_single_line_mode(const scoped_single_line_mode&) = delete;
+ scoped_single_line_mode& operator=(const scoped_single_line_mode&) = delete;
+
+private:
+ fractured_formatter& format_;
+ layout_mode prev_;
};
/**
@@ -11214,10 +12431,12 @@ private:
fractured_json_options options_;
/** Format an element using pre-computed metrics */
- void format_element(const dom::element& elem, const element_metrics& metrics, size_t depth);
+ void format_element(const dom::element& elem, const element_metrics& metrics, size_t depth,
+ bool has_trailing_comma = false);
/** Format an array with the appropriate layout */
- void format_array(const dom::array& arr, const element_metrics& metrics, size_t depth);
+ void format_array(const dom::array& arr, const element_metrics& metrics, size_t depth,
+ bool has_trailing_comma = false);
/** Format an array inline: [1, 2, 3] */
void format_array_inline(const dom::array& arr, const element_metrics& metrics);
@@ -11225,14 +12444,51 @@ private:
/** Format an array with compact multiline: multiple items per line */
void format_array_compact_multiline(const dom::array& arr, const element_metrics& metrics, size_t depth);
+ /** Like format_array_compact_multiline, but rows are cross-row aligned
+ * and packed using a fixed per-row slot width. */
+ void format_array_compact_multiline_aligned(const dom::array& arr, const element_metrics& metrics, size_t depth);
+
/** Format an array as a table */
void format_array_as_table(const dom::array& arr, const element_metrics& metrics, size_t depth);
+ /** Write one object row's columns */
+ void format_table_object_row(const dom::object& obj, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth);
+
+ /** Write one array row's columns */
+ void format_table_array_row(const dom::array& arr, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth);
+
+ /** Dispatches to format_table_object_row/format_table_array_row based on elem's type. */
+ void format_table_row(const dom::element& elem, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth);
+
+ /** Row for a uniform scalar array: writes elem inline, then pads to width so every row lines up. */
+ void format_table_scalar_row(const dom::element& elem, const element_metrics& row_metrics,
+ size_t width, size_t depth, table_column_type column_type);
+
+ /** Shared per-column writer: recurses if the column has children,
+ * otherwise writes a plain padded value or blank. */
+ void format_table_row_columns(const std::vector<table_column>& columns,
+ const std::vector<bool>& found,
+ const std::vector<dom::element>& values,
+ const std::vector<const element_metrics*>& value_metrics,
+ size_t depth);
+
+ /** Writes a single aligned leaf value */
+ void format_table_leaf_value(const dom::element& elem, const element_metrics& vm, size_t width,
+ table_column_type column_type, bool needs_comma,
+ bool add_comma_space, size_t depth);
+
+ /** Whether, for a column of the given type, the comma goes right after the value */
+ bool comma_goes_before_padding(table_column_type column_type) const;
+
/** Format an array expanded: one item per line */
void format_array_expanded(const dom::array& arr, const element_metrics& metrics, size_t depth);
/** Format an object with the appropriate layout */
- void format_object(const dom::object& obj, const element_metrics& metrics, size_t depth);
+ void format_object(const dom::object& obj, const element_metrics& metrics, size_t depth,
+ bool has_trailing_comma = false);
/** Format an object inline: {"a": 1, "b": 2} */
void format_object_inline(const dom::object& obj, const element_metrics& metrics);
@@ -11243,12 +12499,8 @@ private:
/** Format a scalar value */
void format_scalar(const dom::element& elem);
- /** Calculate column widths for table formatting */
- std::vector<size_t> calculate_column_widths(const dom::array& arr,
- const std::vector<std::string>& columns) const;
-
- /** Measure the actual formatted length of a value (for alignment) */
- size_t measure_value_length(const dom::element& elem) const;
+ /** Whether to pad this container's own brackets. */
+ bool bracket_padding_for(const element_metrics& metrics) const;
};
} // namespace internal
@@ -11260,6 +12512,8 @@ private:
#include <cmath>
#include <algorithm>
#include <cstring>
+#include <iterator>
+#include <unordered_map>
namespace simdjson {
namespace internal {
@@ -11290,7 +12544,7 @@ inline element_metrics structure_analyzer::analyze_object(const dom::object& obj
return analyze_object(obj, 0);
}
-inline element_metrics structure_analyzer::analyze_element(const dom::element& elem, size_t depth) {
+inline element_metrics structure_analyzer::analyze_element(const dom::element& elem, size_t depth) const {
switch (elem.type()) {
case dom::element_type::ARRAY: {
dom::array arr;
@@ -11313,12 +12567,11 @@ inline element_metrics structure_analyzer::analyze_element(const dom::element& e
return element_metrics{};
}
-inline element_metrics structure_analyzer::analyze_scalar(const dom::element& elem) {
+inline element_metrics structure_analyzer::analyze_scalar(const dom::element& elem) const {
element_metrics metrics;
metrics.complexity = 0;
metrics.child_count = 0;
metrics.can_inline = true;
- metrics.recommended_layout = layout_mode::INLINE;
switch (elem.type()) {
case dom::element_type::STRING: {
@@ -11367,7 +12620,7 @@ inline element_metrics structure_analyzer::analyze_scalar(const dom::element& el
}
inline element_metrics structure_analyzer::analyze_array(const dom::array& arr,
- size_t depth) {
+ size_t depth) const {
element_metrics metrics;
metrics.complexity = 1; // At least 1 for being an array
metrics.estimated_inline_len = 2; // "[]"
@@ -11392,35 +12645,31 @@ inline element_metrics structure_analyzer::analyze_array(const dom::array& arr,
// Complexity is 1 + max child complexity
metrics.complexity = 1 + max_child_complexity;
- // Check if can inline
- metrics.can_inline = (metrics.complexity <= current_opts_->max_inline_complexity) &&
- (metrics.estimated_inline_len <= current_opts_->max_inline_length);
-
- // Check for uniform array (table formatting)
- if (current_opts_->enable_table_format &&
- metrics.child_count >= current_opts_->min_table_rows) {
- metrics.is_uniform_array = check_array_uniformity(arr, metrics.common_keys);
+ // Bracket padding "[ 1, 2 ]" vs "[1, 2]"
+ bool use_bracket_padding = (max_child_complexity >= 1)
+ ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding;
+ if (use_bracket_padding && metrics.child_count > 0) {
+ metrics.estimated_inline_len += 2;
}
- // Decide layout
- if (metrics.child_count == 0) {
- metrics.recommended_layout = layout_mode::INLINE;
- } else if (metrics.can_inline) {
- metrics.recommended_layout = layout_mode::INLINE;
- } else if (metrics.is_uniform_array && !metrics.common_keys.empty()) {
- metrics.recommended_layout = layout_mode::TABLE;
- } else if (current_opts_->enable_compact_multiline &&
- max_child_complexity <= current_opts_->max_compact_array_complexity) {
- metrics.recommended_layout = layout_mode::COMPACT_MULTILINE;
- } else {
- metrics.recommended_layout = layout_mode::EXPANDED;
+ // Check if can inline
+ metrics.can_inline = metrics.complexity <= current_opts_->max_inline_complexity;
+
+ // Check for uniform array (table formatting, or aligned compact multiline).
+ bool wants_table_columns =
+ (current_opts_->enable_table_format &&
+ max_child_complexity <= current_opts_->max_table_row_complexity) ||
+ (current_opts_->enable_compact_multiline &&
+ max_child_complexity <= current_opts_->max_compact_array_complexity);
+ if (wants_table_columns) {
+ metrics.is_uniform_array = check_array_uniformity(arr, metrics, depth);
}
return metrics;
}
inline element_metrics structure_analyzer::analyze_object(const dom::object& obj,
- size_t depth) {
+ size_t depth) const {
element_metrics metrics;
metrics.complexity = 1;
metrics.estimated_inline_len = 2; // "{}"
@@ -11447,16 +12696,15 @@ inline element_metrics structure_analyzer::analyze_object(const dom::object& obj
metrics.complexity = 1 + max_child_complexity;
- metrics.can_inline = (metrics.complexity <= current_opts_->max_inline_complexity) &&
- (metrics.estimated_inline_len <= current_opts_->max_inline_length);
-
- // Objects use inline or expanded (no table/compact for objects)
- if (metrics.child_count == 0 || metrics.can_inline) {
- metrics.recommended_layout = layout_mode::INLINE;
- } else {
- metrics.recommended_layout = layout_mode::EXPANDED;
+ // Bracket padding '{ "a": 1 }' vs '{"a": 1}'
+ bool use_bracket_padding = (max_child_complexity >= 1)
+ ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding;
+ if (use_bracket_padding && metrics.child_count > 0) {
+ metrics.estimated_inline_len += 2;
}
+ metrics.can_inline = metrics.complexity <= current_opts_->max_inline_complexity;
+
return metrics;
}
@@ -11473,8 +12721,18 @@ inline size_t structure_analyzer::estimate_string_length(std::string_view s) con
}
inline size_t structure_analyzer::estimate_number_length(double d) const {
- if (std::isnan(d) || std::isinf(d)) {
+ if (!std::isfinite(d)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (std::isnan(d)) {
+ return 3; // "NaN"
+ } else if (d < 0) {
+ return 9; // "-Infinity"
+ } else {
+ return 8; // "Infinity"
+ }
+#else
return 4; // "null" for invalid numbers
+#endif
}
// Rough estimate: up to 17 significant digits + sign + decimal point + exponent
char buf[32];
@@ -11505,115 +12763,290 @@ inline size_t structure_analyzer::estimate_number_length(uint64_t u) const {
return len;
}
-inline bool structure_analyzer::check_array_uniformity(const dom::array& arr,
- std::vector<std::string>& common_keys) const {
- common_keys.clear();
-
- std::set<std::string> shared_keys;
- dom::object first_obj;
- bool have_first = false;
- size_t object_count = 0;
+inline table_column_type structure_analyzer::classify_table_value(dom::element_type type) {
+ switch (type) {
+ case dom::element_type::OBJECT: return table_column_type::object;
+ case dom::element_type::ARRAY: return table_column_type::array;
+ case dom::element_type::INT64:
+ case dom::element_type::UINT64:
+ case dom::element_type::DOUBLE: return table_column_type::number;
+ case dom::element_type::NULL_VALUE: return table_column_type::unknown;
+ default: return table_column_type::simple; // string, bool
+ }
+}
- for (dom::element elem : arr) {
- if (elem.type() != dom::element_type::OBJECT) {
- return false; // Not all elements are objects
+inline void structure_analyzer::classify_and_measure(
+ const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+ table_column_type& common, size_t& max_width) {
+ common = table_column_type::unknown;
+ max_width = 0;
+ for (const auto& v : values) {
+ table_column_type t = classify_table_value(v.first.type());
+ if (t != table_column_type::unknown) {
+ if (common == table_column_type::unknown) common = t;
+ else if (t != common) common = table_column_type::mixed;
}
-
- dom::object obj;
- if (elem.get_object().get(obj) != SUCCESS) {
- return false;
+ if (v.second) {
+ max_width = (std::max)(max_width, v.second->estimated_inline_len);
}
+ }
+}
- std::set<std::string> current_keys;
- for (dom::key_value_pair field : obj) {
- current_keys.insert(std::string(field.key));
+inline void structure_analyzer::build_table_columns(
+ const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+ std::vector<table_column>& out_columns) const {
+ out_columns.clear();
+ if (values.empty()) {
+ return;
+ }
+
+ table_column_type common = table_column_type::unknown;
+ for (const auto& v : values) {
+ table_column_type t = classify_table_value(v.first.type());
+ if (t == table_column_type::unknown) continue;
+ if (common == table_column_type::unknown) common = t;
+ else if (t != common) { common = table_column_type::mixed; break; }
+ }
+ if (common != table_column_type::object && common != table_column_type::array) {
+ return;
+ }
+
+ std::vector<std::vector<std::pair<dom::element, const element_metrics*>>> per_column_values;
+
+ if (common == table_column_type::object) {
+ std::unordered_map<std::string_view, size_t> column_index;
+ // Last row (1-based) that filled each column, to detect duplicate keys.
+ std::vector<size_t> column_last_row;
+ size_t row = 0;
+ for (const auto& v : values) {
+ if (v.first.type() != dom::element_type::OBJECT) continue;
+ dom::object obj;
+ if (v.first.get_object().get(obj) != SUCCESS) continue;
+
+ row++;
+ size_t field_idx = 0;
+ for (dom::key_value_pair field : obj) {
+ auto it = column_index.find(field.key);
+ size_t col_idx;
+ if (it == column_index.end()) {
+ col_idx = out_columns.size();
+ column_index.emplace(field.key, col_idx);
+ out_columns.emplace_back();
+ out_columns.back().key.assign(field.key.data(), field.key.size());
+ out_columns.back().has_key = true;
+ out_columns.back().key_width = estimate_string_length(field.key);
+ per_column_values.emplace_back();
+ column_last_row.push_back(0);
+ } else {
+ col_idx = it->second;
+ }
+ // A row with a duplicate key cannot be laid out as a table: each
+ // column holds one value per row, so the later duplicates would be
+ // lost. Returning no columns also disables the aligned compact
+ // multiline layout for this array, which needs the same columns.
+ if (column_last_row[col_idx] == row) {
+ out_columns.clear();
+ return;
+ }
+ column_last_row[col_idx] = row;
+ const element_metrics* field_metrics = (v.second && field_idx < v.second->children.size())
+ ? &v.second->children[field_idx] : nullptr;
+ per_column_values[col_idx].emplace_back(field.value, field_metrics);
+ field_idx++;
+ }
}
+ } else { // array: columns by position
+ for (const auto& v : values) {
+ if (v.first.type() != dom::element_type::ARRAY) continue;
+ dom::array sub_arr;
+ if (v.first.get_array().get(sub_arr) != SUCCESS) continue;
- if (!have_first) {
- shared_keys = current_keys;
- first_obj = obj;
- have_first = true;
- } else {
- // Check similarity threshold against the first object
- double similarity = compute_object_similarity(first_obj, obj);
- if (similarity < current_opts_->table_similarity_threshold) {
- return false; // Objects are too dissimilar for table format
+ size_t idx = 0;
+ for (dom::element item : sub_arr) {
+ if (out_columns.size() <= idx) {
+ out_columns.emplace_back();
+ per_column_values.emplace_back();
+ }
+ const element_metrics* item_metrics = (v.second && idx < v.second->children.size())
+ ? &v.second->children[idx] : nullptr;
+ per_column_values[idx].emplace_back(item, item_metrics);
+ idx++;
}
+ }
+ }
+
+ for (size_t i = 0; i < out_columns.size(); i++) {
+ table_column_type col_type;
+ size_t max_width;
+ classify_and_measure(per_column_values[i], col_type, max_width);
+ out_columns[i].type = col_type;
+ out_columns[i].plain_width = max_width;
- // Intersect with current keys
- std::set<std::string> intersection;
- std::set_intersection(shared_keys.begin(), shared_keys.end(),
- current_keys.begin(), current_keys.end(),
- std::inserter(intersection, intersection.begin()));
- shared_keys = intersection;
+ if (col_type == table_column_type::object || col_type == table_column_type::array) {
+ build_table_columns(per_column_values[i], out_columns[i].children);
}
- object_count++;
+ out_columns[i].width = out_columns[i].children.empty() ? max_width : compute_columns_width(out_columns[i].children);
}
+}
- if (object_count < current_opts_->min_table_rows) {
- return false;
+inline bool structure_analyzer::check_array_uniformity(const dom::array& arr,
+ element_metrics& metrics,
+ size_t depth) const {
+ std::vector<std::pair<dom::element, const element_metrics*>> values;
+ values.reserve(metrics.child_count);
+
+ size_t row_idx = 0;
+ for (dom::element elem : arr) {
+ const element_metrics* row_metrics = (row_idx < metrics.children.size()) ? &metrics.children[row_idx] : nullptr;
+ values.emplace_back(elem, row_metrics);
+ row_idx++;
+ }
+
+ build_table_columns(values, metrics.table_columns);
+ if (metrics.table_columns.empty()) {
+ // Not uniformly object or array. Check for uniform scalar
+ table_column_type common;
+ size_t max_width;
+ classify_and_measure(values, common, max_width);
+ if (common != table_column_type::number && common != table_column_type::simple) {
+ return false;
+ }
+ metrics.scalar_column_type = common;
+ metrics.table_row_width = max_width;
+ metrics.table_row_width_full = max_width;
+ return true;
}
- // Require at least one common key for table formatting
- if (shared_keys.empty()) {
- return false;
+ metrics.table_row_width_full = compute_columns_width(metrics.table_columns);
+
+ size_t row_indent_width = (depth + 1) * current_opts_->indent_spaces;
+ size_t budget = (row_indent_width + 1 >= current_opts_->max_total_line_length)
+ ? 0 : current_opts_->max_total_line_length - row_indent_width - 1;
+
+ size_t width = metrics.table_row_width_full;
+ while (width > budget && flatten_deepest_columns(metrics.table_columns)) {
+ recompute_column_widths(metrics.table_columns);
+ width = compute_columns_width(metrics.table_columns);
}
- common_keys.assign(shared_keys.begin(), shared_keys.end());
+ metrics.table_row_width = width;
return true;
}
-inline double structure_analyzer::compute_object_similarity(const dom::object& a,
- const dom::object& b) const {
- std::set<std::string> keys_a, keys_b;
- for (dom::key_value_pair field : a) {
- keys_a.insert(std::string(field.key));
+inline size_t structure_analyzer::compute_columns_width(const std::vector<table_column>& columns) const {
+ size_t width = 2; // "{}" or "[]"
+ if (table_row_is_nested(columns) ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding) {
+ width += 2;
}
- for (dom::key_value_pair field : b) {
- keys_b.insert(std::string(field.key));
+
+ for (const table_column& col : columns) {
+ if (col.has_key) {
+ width += col.key_width;
+ width += current_opts_->colon_padding ? 2 : 1;
+ }
+ width += col.width;
+ }
+ if (columns.size() > 1) {
+ width += (columns.size() - 1) * (current_opts_->comma_padding ? 2 : 1);
+ }
+ return width;
+}
+
+inline size_t structure_analyzer::column_height(const table_column& column) {
+ size_t height = 0;
+ for (const table_column& child : column.children) {
+ height = (std::max)(height, column_height(child));
}
+ return column.children.empty() ? 0 : height + 1;
+}
- std::set<std::string> intersection;
- std::set_intersection(keys_a.begin(), keys_a.end(),
- keys_b.begin(), keys_b.end(),
- std::inserter(intersection, intersection.begin()));
+inline bool structure_analyzer::flatten_deepest_columns(std::vector<table_column>& columns) {
+ size_t max_height = 0;
+ for (const table_column& col : columns) {
+ max_height = (std::max)(max_height, column_height(col));
+ }
- std::set<std::string> union_set;
- std::set_union(keys_a.begin(), keys_a.end(),
- keys_b.begin(), keys_b.end(),
- std::inserter(union_set, union_set.begin()));
+ bool changed = false;
+ for (table_column& col : columns) {
+ if (column_height(col) != max_height || max_height == 0) continue;
+ if (max_height == 1) {
+ col.children.clear();
+ col.width = col.plain_width;
+ changed = true;
+ } else {
+ changed |= flatten_deepest_columns(col.children);
+ }
+ }
+ return changed;
+}
- if (union_set.empty()) return 1.0;
- return static_cast<double>(intersection.size()) / static_cast<double>(union_set.size());
+inline void structure_analyzer::recompute_column_widths(std::vector<table_column>& columns) const {
+ for (table_column& col : columns) {
+ if (!col.children.empty()) {
+ recompute_column_widths(col.children);
+ col.width = compute_columns_width(col.children);
+ }
+ }
}
inline layout_mode structure_analyzer::decide_layout(const element_metrics& metrics,
size_t depth,
- size_t available_width) const {
+ const fractured_json_options& opts,
+ bool has_trailing_comma) {
if (metrics.child_count == 0) {
- return layout_mode::INLINE;
+ return layout_mode::single_line;
}
+ long long signed_depth = static_cast<long long>(depth);
+ bool depth_allows_inline_or_compact = signed_depth > opts.always_expand_depth;
+ bool depth_allows_table = signed_depth >= opts.always_expand_depth;
+
// Check inline feasibility
- size_t indent_width = depth * current_opts_->indent_spaces;
- if (metrics.can_inline &&
- metrics.estimated_inline_len + indent_width <= available_width) {
- return layout_mode::INLINE;
+ size_t reserved_width = depth * opts.indent_spaces + (has_trailing_comma ? 1 : 0);
+ if (depth_allows_inline_or_compact && metrics.can_inline &&
+ metrics.estimated_inline_len + reserved_width <= opts.max_total_line_length) {
+ return layout_mode::single_line;
}
- // Check table mode
- if (metrics.is_uniform_array && !metrics.common_keys.empty()) {
- return layout_mode::TABLE;
- }
+ // Rows (table's or compact multiline's) render one level deeper than the
+ // array itself.
+ size_t row_indent_width = (depth + 1) * opts.indent_spaces;
// Check compact multiline
- if (current_opts_->enable_compact_multiline &&
- metrics.complexity <= current_opts_->max_compact_array_complexity + 1) {
- return layout_mode::COMPACT_MULTILINE;
+ // for uniform arrays fall back to table if we would have to flatten any formatting
+ bool compact_multiline_enabled = opts.enable_compact_multiline &&
+ metrics.complexity <= opts.max_compact_array_complexity + 1 &&
+ metrics.child_count >= opts.min_compact_array_row_items;
+ if (depth_allows_inline_or_compact && compact_multiline_enabled) {
+ bool aligned = metrics.is_uniform_array;
+ size_t comma_width = opts.comma_padding ? 2 : 1;
+ size_t avg_item_width;
+ if (aligned) {
+ avg_item_width = metrics.table_row_width_full + comma_width;
+ } else {
+ size_t sum = 0;
+ for (const element_metrics& child : metrics.children) {
+ sum += child.estimated_inline_len;
+ }
+ avg_item_width = comma_width + sum / metrics.child_count;
+ }
+
+ size_t row_pack_space = (row_indent_width >= opts.max_total_line_length)
+ ? 0 : opts.max_total_line_length - row_indent_width;
+ if (avg_item_width * opts.min_compact_array_row_items <= row_pack_space) {
+ return layout_mode::compact_multiline;
+ }
+ }
+
+ // Check Table mode
+ if (depth_allows_table && opts.enable_table_format &&
+ metrics.is_uniform_array &&
+ metrics.table_row_width + 1 + row_indent_width <= opts.max_total_line_length) {
+ return layout_mode::table;
}
- return layout_mode::EXPANDED;
+ return layout_mode::expanded;
}
//
@@ -11621,10 +13054,10 @@ inline layout_mode structure_analyzer::decide_layout(const element_metrics& metr
//
inline fractured_formatter::fractured_formatter(const fractured_json_options& opts)
- : options_(opts), column_widths_{} {}
+ : options_(opts) {}
simdjson_inline void fractured_formatter::print_newline() {
- if (current_layout_ == layout_mode::INLINE) {
+ if (current_layout_ == layout_mode::single_line) {
return; // No newlines in inline mode
}
one_char('\n');
@@ -11632,7 +13065,7 @@ simdjson_inline void fractured_formatter::print_newline() {
}
simdjson_inline void fractured_formatter::print_indents(size_t depth) {
- if (current_layout_ == layout_mode::INLINE) {
+ if (current_layout_ == layout_mode::single_line) {
return; // No indentation in inline mode
}
for (size_t i = 0; i < depth * options_.indent_spaces; i++) {
@@ -11654,26 +13087,10 @@ inline layout_mode fractured_formatter::get_layout_mode() const {
return current_layout_;
}
-inline void fractured_formatter::set_depth(size_t depth) {
- current_depth_ = depth;
-}
-
-inline size_t fractured_formatter::get_depth() const {
- return current_depth_;
-}
-
inline void fractured_formatter::track_line_length(size_t chars) {
current_line_length_ += chars;
}
-inline void fractured_formatter::reset_line_length() {
- current_line_length_ = 0;
-}
-
-inline size_t fractured_formatter::get_line_length() const {
- return current_line_length_;
-}
-
inline bool fractured_formatter::should_break_line(size_t upcoming_length) const {
return (current_line_length_ + upcoming_length) > options_.max_total_line_length;
}
@@ -11682,39 +13099,6 @@ inline const fractured_json_options& fractured_formatter::options() const {
return options_;
}
-inline void fractured_formatter::begin_table_row() {
- in_table_mode_ = true;
- current_column_ = 0;
-}
-
-inline void fractured_formatter::end_table_row() {
- in_table_mode_ = false;
- current_column_ = 0;
-}
-
-inline void fractured_formatter::set_column_widths(const std::vector<size_t>& widths) {
- column_widths_ = widths;
-}
-
-inline size_t fractured_formatter::get_column_index() const {
- return current_column_;
-}
-
-inline void fractured_formatter::next_column() {
- current_column_++;
-}
-
-inline void fractured_formatter::align_to_column_width(size_t actual_width) {
- if (current_column_ < column_widths_.size()) {
- size_t target_width = column_widths_[current_column_];
- while (actual_width < target_width) {
- one_char(' ');
- actual_width++;
- current_line_length_++;
- }
- }
-}
-
//
// Fractured String Builder Implementation
//
@@ -11753,19 +13137,20 @@ simdjson_inline std::string_view fractured_string_builder::str() const {
inline void fractured_string_builder::format_element(const dom::element& elem,
const element_metrics& metrics,
- size_t depth) {
+ size_t depth,
+ bool has_trailing_comma) {
switch (elem.type()) {
case dom::element_type::ARRAY: {
dom::array arr;
if (elem.get_array().get(arr) == SUCCESS) {
- format_array(arr, metrics, depth);
+ format_array(arr, metrics, depth, has_trailing_comma);
}
break;
}
case dom::element_type::OBJECT: {
dom::object obj;
if (elem.get_object().get(obj) == SUCCESS) {
- format_object(obj, metrics, depth);
+ format_object(obj, metrics, depth, has_trailing_comma);
}
break;
}
@@ -11777,18 +13162,20 @@ inline void fractured_string_builder::format_element(const dom::element& elem,
inline void fractured_string_builder::format_array(const dom::array& arr,
const element_metrics& metrics,
- size_t depth) {
- switch (metrics.recommended_layout) {
- case layout_mode::INLINE:
+ size_t depth,
+ bool has_trailing_comma) {
+ layout_mode layout = structure_analyzer::decide_layout(metrics, depth, options_, has_trailing_comma);
+ switch (layout) {
+ case layout_mode::single_line:
format_array_inline(arr, metrics);
break;
- case layout_mode::COMPACT_MULTILINE:
+ case layout_mode::compact_multiline:
format_array_compact_multiline(arr, metrics, depth);
break;
- case layout_mode::TABLE:
+ case layout_mode::table:
format_array_as_table(arr, metrics, depth);
break;
- case layout_mode::EXPANDED:
+ case layout_mode::expanded:
default:
format_array_expanded(arr, metrics, depth);
break;
@@ -11797,8 +13184,7 @@ inline void fractured_string_builder::format_array(const dom::array& arr,
inline void fractured_string_builder::format_array_inline(const dom::array& arr,
const element_metrics& metrics) {
- layout_mode prev_layout = format_.get_layout_mode();
- format_.set_layout_mode(layout_mode::INLINE);
+ scoped_single_line_mode single_line(format_);
format_.start_array();
@@ -11812,60 +13198,69 @@ inline void fractured_string_builder::format_array_inline(const dom::array& arr,
if (options_.comma_padding) {
format_.print_space();
}
- } else if (options_.simple_bracket_padding) {
+ } else if (bracket_padding_for(metrics)) {
format_.print_space();
}
first = false;
- const element_metrics& child_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
+ const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
format_element(elem, child_metrics, 0);
child_idx++;
}
- if (options_.simple_bracket_padding && !empty) {
+ if (bracket_padding_for(metrics) && !empty) {
format_.print_space();
}
format_.end_array();
-
- format_.set_layout_mode(prev_layout);
}
inline void fractured_string_builder::format_array_compact_multiline(const dom::array& arr,
const element_metrics& metrics,
size_t depth) {
+ if (metrics.is_uniform_array) {
+ format_array_compact_multiline_aligned(arr, metrics, depth);
+ return;
+ }
+
format_.start_array();
format_.print_newline();
format_.print_indents(depth + 1);
- size_t items_on_line = 0;
bool first = true;
+ bool prev_item_was_expanded = false;
size_t child_idx = 0;
for (dom::element elem : arr) {
+ const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+
if (!first) {
format_.comma();
+ format_.track_line_length(1);
// Check if we should break to new line
- if (items_on_line >= options_.max_items_per_line ||
- format_.should_break_line(20)) { // 20 is rough estimate for next item
+ if (prev_item_was_expanded ||
+ format_.should_break_line(child_metrics.estimated_inline_len)) {
format_.print_newline();
format_.print_indents(depth + 1);
- items_on_line = 0;
} else if (options_.comma_padding) {
format_.print_space();
}
}
first = false;
- // Format element inline
- layout_mode prev_layout = format_.get_layout_mode();
- format_.set_layout_mode(layout_mode::INLINE);
- const element_metrics& child_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
- format_element(elem, child_metrics, depth + 1);
- format_.set_layout_mode(prev_layout);
+ bool is_last = (child_idx + 1 == metrics.child_count);
+ layout_mode item_layout = structure_analyzer::decide_layout(child_metrics, depth + 1, options_, !is_last);
+ bool item_fits = item_layout == layout_mode::single_line;
+ if (item_fits) {
+ {
+ scoped_single_line_mode single_line(format_);
+ format_element(elem, child_metrics, depth + 1, !is_last);
+ }
+ format_.track_line_length(child_metrics.estimated_inline_len);
+ } else {
+ format_element(elem, child_metrics, depth + 1, !is_last);
+ }
+ prev_item_was_expanded = !item_fits;
- items_on_line++;
child_idx++;
}
@@ -11874,114 +13269,309 @@ inline void fractured_string_builder::format_array_compact_multiline(const dom::
format_.end_array();
}
-inline void fractured_string_builder::format_array_as_table(const dom::array& arr,
- const element_metrics& metrics,
- size_t depth) {
- const std::vector<std::string>& columns = metrics.common_keys;
- if (columns.empty()) {
- format_array_expanded(arr, metrics, depth);
- return;
- }
-
- // Calculate column widths for alignment
- std::vector<size_t> col_widths = calculate_column_widths(arr, columns);
- format_.set_column_widths(col_widths);
+inline void fractured_string_builder::format_array_compact_multiline_aligned(
+ const dom::array& arr, const element_metrics& metrics, size_t depth) {
+ const std::vector<table_column>& columns = metrics.table_columns;
format_.start_array();
format_.print_newline();
+ format_.print_indents(depth + 1);
- bool first_row = true;
+ size_t indent_width = (depth + 1) * options_.indent_spaces;
+ size_t available_line_space = (indent_width >= options_.max_total_line_length)
+ ? 0 : options_.max_total_line_length - indent_width;
+ size_t comma_width = options_.comma_padding ? 2 : 1;
+ size_t remaining_line_space = available_line_space;
+
+ bool first = true;
size_t child_idx = 0;
+
for (dom::element elem : arr) {
- if (!first_row) {
- format_.comma();
- format_.print_newline();
- }
- first_row = false;
+ bool needs_comma = (child_idx + 1 < metrics.child_count);
+ size_t space_needed = metrics.table_row_width_full + (needs_comma ? comma_width : 0);
- format_.print_indents(depth + 1);
- format_.begin_table_row();
+ if (!first) {
+ if (remaining_line_space < space_needed) {
+ format_.print_newline();
+ format_.print_indents(depth + 1);
+ remaining_line_space = available_line_space;
+ } else if (options_.comma_padding) {
+ format_.print_space();
+ }
+ }
+ first = false;
- // Format object as inline with aligned columns
- dom::object obj;
- if (elem.get_object().get(obj) != SUCCESS) {
- child_idx++;
- continue;
+ const element_metrics& row_metrics = child_metrics_at(metrics.children, child_idx);
+ if (columns.empty()) {
+ format_table_scalar_row(elem, row_metrics, metrics.table_row_width_full, depth + 1,
+ metrics.scalar_column_type);
+ } else {
+ format_table_row(elem, row_metrics, columns, depth + 1);
}
+ if (needs_comma) {
+ format_.comma();
+ }
+ remaining_line_space -= (std::min)(remaining_line_space, space_needed);
+ child_idx++;
+ }
- // Get child metrics for this row (object)
- const element_metrics& row_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
+ format_.print_newline();
+ format_.print_indents(depth);
+ format_.end_array();
+}
- format_.start_object();
- if (options_.simple_bracket_padding) {
- format_.print_space();
- }
+inline void fractured_string_builder::format_table_row_columns(
+ const std::vector<table_column>& columns,
+ const std::vector<bool>& found,
+ const std::vector<dom::element>& values,
+ const std::vector<const element_metrics*>& value_metrics,
+ size_t depth) {
+ const size_t num_columns = columns.size();
+ size_t last_present_idx = num_columns;
+ for (size_t i = 0; i < num_columns; i++) {
+ if (found[i]) last_present_idx = i;
+ }
- bool first_col = true;
- const size_t num_columns = columns.size();
+ size_t comma_width = options_.comma_padding ? 2 : 1;
- for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
- const std::string& key = columns[col_idx];
- const bool is_last_col = (col_idx == num_columns - 1);
+ for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
+ const table_column& column = columns[col_idx];
+ const bool is_last_col = (col_idx == num_columns - 1);
- if (!first_col) {
- format_.comma();
- if (options_.comma_padding) {
+ if (found[col_idx]) {
+ if (column.has_key) {
+ format_.key(column.key);
+ if (options_.colon_padding) {
format_.print_space();
}
}
- first_col = false;
- // Write key
- format_.key(key);
- if (options_.colon_padding) {
- format_.print_space();
- }
+ bool needs_comma = !is_last_col && (col_idx < last_present_idx);
- // Find the value for this key and its metrics
- dom::element value;
- bool found = false;
- size_t field_idx = 0;
- for (dom::key_value_pair field : obj) {
- if (field.key == key) {
- value = field.value;
- found = true;
- break;
+ if (!column.children.empty()) {
+ // Recurses into this cell's own columns instead of a plain value;
+ // every row aligns those the same way (blank-padding missing
+ // ones), so the result is always exactly column.width wide
+ // no padding needed afterward, unlike the leaf case below.
+ if (column.type == table_column_type::object) {
+ dom::object sub_obj;
+ if (values[col_idx].get_object().get(sub_obj) == SUCCESS) {
+ const element_metrics& sub_metrics = child_metrics_at(value_metrics[col_idx]);
+ format_table_object_row(sub_obj, sub_metrics, column.children, depth);
+ }
+ } else {
+ dom::array sub_arr;
+ if (values[col_idx].get_array().get(sub_arr) == SUCCESS) {
+ const element_metrics& sub_metrics = child_metrics_at(value_metrics[col_idx]);
+ format_table_array_row(sub_arr, sub_metrics, column.children, depth);
+ }
}
- field_idx++;
+ // value is already padded
+ if (needs_comma) {
+ format_.comma();
+ if (options_.comma_padding) {
+ format_.print_space();
+ }
+ }
+ } else {
+ const element_metrics& vm = child_metrics_at(value_metrics[col_idx]);
+ format_table_leaf_value(values[col_idx], vm, column.width, column.type, needs_comma,
+ /*add_comma_space=*/true, depth);
}
- // Write value
- if (found) {
- layout_mode prev_layout = format_.get_layout_mode();
- format_.set_layout_mode(layout_mode::INLINE);
- const element_metrics& value_metrics = (field_idx < row_metrics.children.size())
- ? row_metrics.children[field_idx] : element_metrics{};
- format_element(value, value_metrics, depth + 1);
- format_.set_layout_mode(prev_layout);
- } else {
- format_.null_atom();
+ if (!is_last_col && !needs_comma) {
+ // Found, but no more real values follow: blank space where a comma would go.
+ for (size_t i = 0; i < comma_width; i++) {
+ format_.one_char(' ');
+ }
+ }
+ } else {
+ size_t slot_width = column.width;
+ if (column.has_key) {
+ slot_width += column.key_width + (options_.colon_padding ? 2 : 1);
+ }
+ for (size_t i = 0; i < slot_width; i++) {
+ format_.one_char(' ');
}
- // Only pad non-last columns to align values across rows
if (!is_last_col) {
- size_t actual_len = found ? measure_value_length(value) : 4; // 4 for "null"
- size_t target_width = col_widths[col_idx];
- while (actual_len < target_width) {
+ for (size_t i = 0; i < comma_width; i++) {
format_.one_char(' ');
- actual_len++;
}
}
+ }
+ }
+}
- format_.next_column();
+inline void fractured_string_builder::format_table_object_row(
+ const dom::object& obj, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth) {
+ const size_t num_columns = columns.size();
+ std::vector<bool> found(num_columns, false);
+ std::vector<dom::element> values(num_columns);
+ std::vector<const element_metrics*> value_metrics(num_columns, nullptr);
+
+ for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
+ size_t field_idx = 0;
+ for (dom::key_value_pair field : obj) {
+ if (field.key == columns[col_idx].key) {
+ found[col_idx] = true;
+ values[col_idx] = field.value;
+ value_metrics[col_idx] = (field_idx < row_metrics.children.size())
+ ? &row_metrics.children[field_idx] : nullptr;
+ break;
+ }
+ field_idx++;
}
+ }
- if (options_.simple_bracket_padding) {
- format_.print_space();
+ bool nested = table_row_is_nested(columns);
+ format_.start_object();
+ if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+ format_.print_space();
+ }
+ format_table_row_columns(columns, found, values, value_metrics, depth);
+ if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+ format_.print_space();
+ }
+ format_.end_object();
+}
+
+inline void fractured_string_builder::format_table_array_row(
+ const dom::array& arr, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth) {
+ const size_t num_columns = columns.size();
+ std::vector<bool> found(num_columns, false);
+ std::vector<dom::element> values(num_columns);
+ std::vector<const element_metrics*> value_metrics(num_columns, nullptr);
+
+ size_t idx = 0;
+ for (dom::element item : arr) {
+ if (idx >= num_columns) break;
+ found[idx] = true;
+ values[idx] = item;
+ value_metrics[idx] = (idx < row_metrics.children.size()) ? &row_metrics.children[idx] : nullptr;
+ idx++;
+ }
+
+ bool nested = table_row_is_nested(columns);
+ format_.start_array();
+ if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+ format_.print_space();
+ }
+ format_table_row_columns(columns, found, values, value_metrics, depth);
+ if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+ format_.print_space();
+ }
+ format_.end_array();
+}
+
+inline void fractured_string_builder::format_table_row(
+ const dom::element& elem, const element_metrics& row_metrics,
+ const std::vector<table_column>& columns, size_t depth) {
+ if (elem.type() == dom::element_type::ARRAY) {
+ dom::array arr;
+ if (elem.get_array().get(arr) == SUCCESS) {
+ format_table_array_row(arr, row_metrics, columns, depth);
+ }
+ } else {
+ dom::object obj;
+ if (elem.get_object().get(obj) == SUCCESS) {
+ format_table_object_row(obj, row_metrics, columns, depth);
+ }
+ }
+}
+
+inline bool fractured_string_builder::comma_goes_before_padding(table_column_type column_type) const {
+ switch (options_.comma_placement) {
+ case table_comma_placement::before_padding: return true;
+ case table_comma_placement::after_padding: return false;
+ case table_comma_placement::before_padding_except_numbers:
+ default:
+ return column_type != table_column_type::number;
+ }
+}
+
+inline void fractured_string_builder::format_table_leaf_value(
+ const dom::element& elem, const element_metrics& vm, size_t width,
+ table_column_type column_type, bool needs_comma, bool add_comma_space, size_t depth) {
+ bool comma_before_pad = needs_comma && comma_goes_before_padding(column_type);
+ bool comma_after_pad = needs_comma && !comma_before_pad;
+
+ bool right_align = column_type == table_column_type::number &&
+ options_.number_alignment == number_list_alignment::right;
+
+ size_t value_len = vm.estimated_inline_len;
+ size_t left_pad = 0;
+ size_t right_pad = 0;
+ if (right_align) {
+ left_pad = (width > value_len) ? width - value_len : 0;
+ comma_before_pad = needs_comma;
+ comma_after_pad = false;
+ } else {
+ right_pad = (width > value_len) ? width - value_len : 0;
+ }
+
+ for (size_t i = 0; i < left_pad; i++) {
+ format_.one_char(' ');
+ }
+
+ {
+ scoped_single_line_mode single_line(format_);
+ format_element(elem, vm, depth);
+ }
+
+ if (comma_before_pad) {
+ format_.comma();
+ }
+ for (size_t i = 0; i < right_pad; i++) {
+ format_.one_char(' ');
+ }
+ if (comma_after_pad) {
+ format_.comma();
+ }
+ if (needs_comma && add_comma_space && options_.comma_padding) {
+ format_.print_space();
+ }
+}
+
+inline void fractured_string_builder::format_table_scalar_row(
+ const dom::element& elem, const element_metrics& row_metrics, size_t width, size_t depth,
+ table_column_type column_type) {
+ format_table_leaf_value(elem, row_metrics, width, column_type,
+ /*needs_comma=*/false, /*add_comma_space=*/false, depth);
+}
+
+inline void fractured_string_builder::format_array_as_table(const dom::array& arr,
+ const element_metrics& metrics,
+ size_t depth) {
+ if (!metrics.is_uniform_array) {
+ format_array_expanded(arr, metrics, depth);
+ return;
+ }
+ const std::vector<table_column>& columns = metrics.table_columns;
+
+ format_.start_array();
+ format_.print_newline();
+
+ bool first_row = true;
+ size_t child_idx = 0;
+ for (dom::element elem : arr) {
+ if (!first_row) {
+ format_.comma();
+ format_.print_newline();
+ }
+ first_row = false;
+
+ format_.print_indents(depth + 1);
+
+ const element_metrics& row_metrics = child_metrics_at(metrics.children, child_idx);
+ if (columns.empty()) {
+ format_table_scalar_row(elem, row_metrics, metrics.table_row_width, depth + 1,
+ metrics.scalar_column_type);
+ } else {
+ format_table_row(elem, row_metrics, columns, depth + 1);
}
- format_.end_object();
- format_.end_table_row();
child_idx++;
}
@@ -12008,9 +13598,9 @@ inline void fractured_string_builder::format_array_expanded(const dom::array& ar
format_.print_newline();
format_.print_indents(depth + 1);
- const element_metrics& child_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
- format_element(elem, child_metrics, depth + 1);
+ const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+ bool is_last = (child_idx + 1 == metrics.child_count);
+ format_element(elem, child_metrics, depth + 1, !is_last);
child_idx++;
}
@@ -12023,8 +13613,10 @@ inline void fractured_string_builder::format_array_expanded(const dom::array& ar
inline void fractured_string_builder::format_object(const dom::object& obj,
const element_metrics& metrics,
- size_t depth) {
- if (metrics.recommended_layout == layout_mode::INLINE || metrics.can_inline) {
+ size_t depth,
+ bool has_trailing_comma) {
+ layout_mode layout = structure_analyzer::decide_layout(metrics, depth, options_, has_trailing_comma);
+ if (layout == layout_mode::single_line) {
format_object_inline(obj, metrics);
} else {
format_object_expanded(obj, metrics, depth);
@@ -12033,8 +13625,7 @@ inline void fractured_string_builder::format_object(const dom::object& obj,
inline void fractured_string_builder::format_object_inline(const dom::object& obj,
const element_metrics& metrics) {
- layout_mode prev_layout = format_.get_layout_mode();
- format_.set_layout_mode(layout_mode::INLINE);
+ scoped_single_line_mode single_line(format_);
format_.start_object();
@@ -12049,7 +13640,7 @@ inline void fractured_string_builder::format_object_inline(const dom::object& ob
if (options_.comma_padding) {
format_.print_space();
}
- } else if (options_.simple_bracket_padding) {
+ } else if (bracket_padding_for(metrics)) {
format_.print_space();
}
first = false;
@@ -12058,18 +13649,15 @@ inline void fractured_string_builder::format_object_inline(const dom::object& ob
if (options_.colon_padding) {
format_.print_space();
}
- const element_metrics& child_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
+ const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
format_element(field.value, child_metrics, 0);
child_idx++;
}
- if (options_.simple_bracket_padding && !empty) {
+ if (bracket_padding_for(metrics) && !empty) {
format_.print_space();
}
format_.end_object();
-
- format_.set_layout_mode(prev_layout);
}
inline void fractured_string_builder::format_object_expanded(const dom::object& obj,
@@ -12094,9 +13682,9 @@ inline void fractured_string_builder::format_object_expanded(const dom::object&
if (options_.colon_padding) {
format_.print_space();
}
- const element_metrics& child_metrics = (child_idx < metrics.children.size())
- ? metrics.children[child_idx] : element_metrics{};
- format_element(field.value, child_metrics, depth + 1);
+ const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+ bool is_last = (child_idx + 1 == metrics.child_count);
+ format_element(field.value, child_metrics, depth + 1, !is_last);
child_idx++;
}
@@ -12152,97 +13740,8 @@ inline void fractured_string_builder::format_scalar(const dom::element& elem) {
}
}
-inline size_t fractured_string_builder::measure_value_length(const dom::element& elem) const {
- switch (elem.type()) {
- case dom::element_type::STRING: {
- std::string_view str;
- if (elem.get_string().get(str) == SUCCESS) {
- // Count actual escaped length
- size_t len = 2; // quotes
- for (char c : str) {
- if (c == '"' || c == '\\' || static_cast<unsigned char>(c) < 32) {
- len += 2; // escape sequence
- } else {
- len += 1;
- }
- }
- return len;
- }
- return 2;
- }
- case dom::element_type::INT64: {
- int64_t val;
- if (elem.get_int64().get(val) == SUCCESS) {
- if (val == 0) return 1;
- // Handle INT64_MIN specially to avoid overflow when negating
- if (val == INT64_MIN) return 20; // "-9223372036854775808" is 20 characters
- size_t len = (val < 0) ? 1 : 0;
- int64_t abs_val = (val < 0) ? -val : val;
- while (abs_val > 0) { len++; abs_val /= 10; }
- return len;
- }
- return 1;
- }
- case dom::element_type::UINT64: {
- uint64_t val;
- if (elem.get_uint64().get(val) == SUCCESS) {
- if (val == 0) return 1;
- size_t len = 0;
- while (val > 0) { len++; val /= 10; }
- return len;
- }
- return 1;
- }
- case dom::element_type::DOUBLE: {
- double val;
- if (elem.get_double().get(val) == SUCCESS) {
- char buf[32];
- int len = snprintf(buf, sizeof(buf), "%.17g", val);
- return len > 0 ? static_cast<size_t>(len) : 1;
- }
- return 1;
- }
- case dom::element_type::BOOL: {
- bool val;
- if (elem.get_bool().get(val) == SUCCESS) {
- return val ? 4 : 5; // "true" or "false"
- }
- return 5;
- }
- case dom::element_type::NULL_VALUE:
- return 4; // "null"
- default:
- return 4;
- }
-}
-
-inline std::vector<size_t> fractured_string_builder::calculate_column_widths(
- const dom::array& arr,
- const std::vector<std::string>& columns) const {
-
- std::vector<size_t> widths(columns.size(), 0);
-
- for (dom::element elem : arr) {
- dom::object obj;
- if (elem.get_object().get(obj) != SUCCESS) {
- continue;
- }
-
- for (size_t col_idx = 0; col_idx < columns.size(); col_idx++) {
- const std::string& key = columns[col_idx];
-
- for (dom::key_value_pair field : obj) {
- if (field.key == key) {
- // Measure actual value length
- size_t len = measure_value_length(field.value);
- widths[col_idx] = (std::max)(widths[col_idx], len);
- break;
- }
- }
- }
- }
-
- return widths;
+inline bool fractured_string_builder::bracket_padding_for(const element_metrics& metrics) const {
+ return metrics.complexity >= 2 ? options_.nested_bracket_padding : options_.simple_bracket_padding;
}
} // namespace internal
@@ -12282,19 +13781,6 @@ std::string fractured_json(simdjson_result<T> x, const fractured_json_options& o
}
#endif
-// Explicit template instantiations for common types
-template std::string fractured_json(dom::element x);
-template std::string fractured_json(dom::element x, const fractured_json_options& options);
-template std::string fractured_json(dom::array x);
-template std::string fractured_json(dom::array x, const fractured_json_options& options);
-template std::string fractured_json(dom::object x);
-template std::string fractured_json(dom::object x, const fractured_json_options& options);
-
-#if SIMDJSON_EXCEPTIONS
-template std::string fractured_json(simdjson_result<dom::element> x);
-template std::string fractured_json(simdjson_result<dom::element> x, const fractured_json_options& options);
-#endif
-
//
// String-based API for formatting any JSON string
//
@@ -12662,6 +14148,8 @@ enum instruction_set {
LASX = 0x40000,
//RVV = 0x80000,
RVV_VLS = 0x100000,
+ SVE = 0x200000,
+ SVE2 = 0x400000,
};
} // namespace internal
@@ -13774,7 +15262,7 @@ public:
simdjson_inline implementation() : simdjson::implementation(
"rvv_vls",
"RISC-V V extension",
- 0
+ internal::instruction_set::RVV_VLS
) {}
simdjson_warn_unused error_code create_dom_parser_implementation(
size_t capacity,
@@ -13933,7 +15421,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {
/* result might be undefined when input_num is zero */
simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+ // if the system supports SVE or CSSC, __builtin_popcountll
+ // might be compiled to fewer single instructions. For CSSC,
+ // __builtin_popcountll is compiled to a single instruction.
+ return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
}
@@ -13970,15 +15465,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace arm64
@@ -14242,6 +15728,7 @@ namespace {
return vget_lane_u64(
vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
}
+ // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
};
@@ -14890,6 +16377,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -14901,6 +16391,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -14937,6 +16449,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace arm64
@@ -15027,7 +16604,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -15206,6 +16783,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -15245,6 +16823,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -15501,6 +17091,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -15538,6 +17341,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -15556,6 +17369,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -15572,26 +17407,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -15680,7 +17562,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -15763,15 +17645,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -15802,7 +17686,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -15851,7 +17749,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -15950,7 +17848,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -16048,7 +17946,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -16103,7 +18001,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -16189,7 +18087,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -16229,11 +18127,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -16244,9 +18151,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -16295,67 +18201,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
//
// Check for minus sign
//
@@ -16367,91 +18218,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+ if ( p == src ) {
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
- }
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
-
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
+ return INCORRECT_TYPE;
}
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -16462,9 +18242,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
+ p += parse_digit(*p, i);
+ bool leading_zero = (i == 0);
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) { return INCORRECT_TYPE; }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ p++;
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -16513,6 +18523,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -17109,6 +19212,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -17120,6 +19226,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -17156,6 +19284,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace fallback
@@ -17246,7 +19439,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -17425,6 +19618,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -17464,6 +19658,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -17720,6 +19926,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -17757,6 +20176,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -17775,6 +20204,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -17791,26 +20242,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -17899,7 +20397,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -17982,15 +20480,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -18021,7 +20521,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -18070,7 +20584,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -18169,7 +20683,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -18267,7 +20781,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -18322,7 +20836,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -18408,7 +20922,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -18448,11 +20962,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -18463,9 +20986,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18514,6 +21036,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -18666,11 +21279,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -18681,9 +21308,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18732,6 +21358,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -19053,16 +21772,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace haswell
@@ -19815,6 +22524,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -19826,6 +22538,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -19862,6 +22596,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace haswell
@@ -19952,7 +22751,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -20131,6 +22930,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -20170,6 +22970,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -20426,6 +23238,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -20463,6 +23488,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -20481,6 +23516,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -20497,26 +23554,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -20605,7 +23709,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -20688,15 +23792,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -20727,7 +23833,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -20776,7 +23896,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -20875,7 +23995,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -20973,7 +24093,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -21028,7 +24148,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -21114,7 +24234,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -21154,11 +24274,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -21169,9 +24298,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -21220,6 +24348,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -21372,11 +24591,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -21387,9 +24620,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -21438,6 +24670,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -21756,16 +25081,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace icelake
@@ -22521,6 +25836,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -22532,6 +25850,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -22568,6 +25908,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace icelake
@@ -22658,7 +26063,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -22837,6 +26242,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -22876,6 +26282,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -23132,6 +26550,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -23169,6 +26800,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -23187,6 +26828,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -23203,26 +26866,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -23311,7 +27021,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -23394,15 +27104,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -23433,7 +27145,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -23482,7 +27208,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -23581,7 +27307,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -23679,7 +27405,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -23734,7 +27460,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -23820,7 +27546,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -23860,11 +27586,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -23875,9 +27610,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -23926,6 +27660,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -24078,11 +27903,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -24093,9 +27932,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24144,6 +27982,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -24434,16 +28365,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace ppc64
@@ -25342,6 +29263,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -25353,6 +29277,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -25389,6 +29335,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace ppc64
@@ -25479,7 +29490,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -25658,6 +29669,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -25697,6 +29709,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -25953,6 +29977,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -25990,6 +30227,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -26008,6 +30255,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -26024,26 +30293,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -26132,7 +30448,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -26215,15 +30531,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -26254,7 +30572,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -26303,7 +30635,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -26402,7 +30734,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -26500,7 +30832,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -26555,7 +30887,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -26641,7 +30973,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -26681,11 +31013,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -26696,9 +31037,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -26747,67 +31087,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
//
// Check for minus sign
//
@@ -26819,91 +31104,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+ if ( p == src ) {
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
- }
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
-
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
+ return INCORRECT_TYPE;
}
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -26914,9 +31128,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
+ p += parse_digit(*p, i);
+ bool leading_zero = (i == 0);
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) { return INCORRECT_TYPE; }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ p++;
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -26965,6 +31409,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -27269,16 +31806,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -27857,16 +32384,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -28480,6 +32997,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -28491,6 +33011,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -28527,6 +33069,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace westmere
@@ -28617,7 +33224,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -28796,6 +33403,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -28835,6 +33443,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -29091,6 +33711,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -29128,6 +33961,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -29146,6 +33989,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -29162,26 +34027,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -29270,7 +34182,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -29353,15 +34265,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -29392,7 +34306,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -29441,7 +34369,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -29540,7 +34468,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -29638,7 +34566,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -29693,7 +34621,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -29779,7 +34707,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -29819,11 +34747,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -29834,9 +34771,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -29885,6 +34821,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -30037,11 +35064,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -30052,9 +35093,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -30103,6 +35143,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -30368,10 +35501,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lasx
@@ -31118,6 +36247,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -31129,6 +36261,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -31165,6 +36319,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace lasx
@@ -31255,7 +36474,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -31434,6 +36653,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -31473,6 +36693,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -31729,6 +36961,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -31766,6 +37211,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -31784,6 +37239,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -31800,26 +37277,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -31908,7 +37432,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -31991,15 +37515,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -32030,7 +37556,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -32079,7 +37619,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -32178,7 +37718,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -32276,7 +37816,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -32331,7 +37871,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -32417,7 +37957,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -32457,11 +37997,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -32472,9 +38021,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -32523,6 +38071,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -32675,11 +38314,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -32690,9 +38343,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -32741,6 +38393,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -33002,10 +38747,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lsx
@@ -33734,6 +39475,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -33745,6 +39489,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -33781,6 +39547,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace lsx
@@ -33871,7 +39702,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -34050,6 +39881,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -34089,6 +39921,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -34345,6 +40189,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -34382,6 +40439,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -34400,6 +40467,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -34416,26 +40505,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -34524,7 +40660,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -34607,15 +40743,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -34646,7 +40784,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -34695,7 +40847,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -34794,7 +40946,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -34892,7 +41044,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -34947,7 +41099,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -35033,7 +41185,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -35073,11 +41225,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -35088,9 +41249,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -35139,6 +41299,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
return (*src == '-');
}
@@ -35291,11 +41542,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -35306,9 +41571,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -35357,6 +41621,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -35622,11 +41979,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
return __builtin_popcountll(input_num);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace rvv_vls
@@ -36367,6 +42719,9 @@ namespace atomparsing {
// to the compile-time constant 1936482662.
simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }
+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+
// Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
// Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -36378,6 +42733,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
return srcval ^ string_to_uint32(atom);
}
+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+ static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+ std::memcpy(&srcval, src, sizeof(uint64_t));
+
+ return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+ return ((src[0] | 0x20) ^ atom[0]) //
+ | ((src[1] | 0x20) ^ atom[1]) //
+ | ((src[2] | 0x20) ^ atom[2]);
+}
+
simdjson_warn_unused
simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -36414,6 +42791,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
else { return false; }
}
+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+ return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+ if (len > 3) { return is_valid_nan_atom(src); }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+ return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ if(is_short_inf) return true;
+
+ // Check for 'infinity' (any capitalization)
+ return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+ bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+ if(is_short_inf) return true;
+
+ return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+ if (len > 8) { return is_valid_inf_atom(src); }
+ if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+ return true;
+ }
+ if (len > 3) {
+ return (str3ncmp_case_insensitive(src, "inf")
+ | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+ }
+ if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+ return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
} // namespace atomparsing
} // unnamed namespace
} // namespace rvv_vls
@@ -36504,7 +42946,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
}
inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
- if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+ if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
// Stage 2 stacks
open_containers.reset(new (std::nothrow) open_container[max_depth]);
is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -36683,6 +43125,7 @@ protected:
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
/* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
@@ -36722,6 +43165,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
return d;
}
+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+ float f;
+ mantissa &= ~(uint32_t(1) << 23);
+ mantissa |= real_exponent << 23;
+ mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+ std::memcpy(&f, &mantissa, sizeof(f));
+ return f;
+}
+
// Attempts to compute i * 10^(power) exactly; and if "negative" is
// true, negate the result.
// This function will only work in some cases, when it does not work, success is
@@ -36978,6 +43433,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
return true;
}
+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+ // Powers of ten that are exactly representable as binary32 values: 10^k is
+ // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+ static constexpr float power_of_ten_float[] = {
+ 1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+ // The range of powers of ten that a non-zero, finite binary32 value can be
+ // built from. Anything smaller rounds to zero, anything larger is infinite.
+ // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+ // fast_float uses for binary32.
+ constexpr int smallest_power_binary32 = -65;
+ constexpr int largest_power_binary32 = 38;
+
+ // We start with the fast path described in
+ // Clinger WD. How to read floating point numbers accurately.
+ // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ // We cannot be certain that x/y is rounded to nearest.
+ if (0 <= power && power <= 10 && i <= 16777215)
+#else
+ if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+ {
+ // Convert the integer into a float. This is lossless since
+ // 0 <= i <= 2^24 - 1.
+ d = float(i);
+ // Both d and the power of ten are exactly representable as binary32
+ // values, so the product (or quotient) is correctly rounded.
+ if (power < 0) {
+ d = d / power_of_ten_float[-power];
+ } else {
+ d = d * power_of_ten_float[power];
+ }
+ if (negative) {
+ d = -d;
+ }
+ return true;
+ }
+
+ // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+ // It needs i > 0 (so that the leading bit of i can be normalized), so we
+ // handle i == 0 separately. We also handle the powers of ten that are so
+ // small (or so large) that the answer is zero (or infinite) whatever the
+ // mantissa is.
+ if (i == 0 || power < smallest_power_binary32) {
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ if (power > largest_power_binary32) {
+ // We have, for sure, an infinite value.
+ return false;
+ }
+
+ // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+ // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+ // The 63 comes from the fact that we use a 64-bit word.
+ // See compute_float_64 for a discussion of the magical 152170 + 65536.
+ int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(i);
+ i <<= lz;
+
+ // We want the most significant 64 bits of the product i * 5**power. It is
+ // safe to index the table because
+ // smallest_power <= smallest_power_binary32 <= power
+ // <= largest_power_binary32 <= largest_power.
+ const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+ simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+ // Unless the least significant 38 bits of the high (64-bit) part of the full
+ // product are all 1s, then we know that the most significant 26 bits are
+ // exact and no further work is needed. Having 26 bits is necessary because
+ // we need 24 bits for the mantissa but we have to have one rounding bit and
+ // we can waste a bit if the most significant bit of the product is zero.
+ // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+ if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+ // The truncated multiplication was not accurate enough; use the next 64
+ // bits of the power of five to refine it. See compute_float_64 for a
+ // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+ simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+ firstproduct.low += secondproduct.high;
+ if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+ }
+ uint64_t lower = firstproduct.low;
+ uint64_t upper = firstproduct.high;
+ // The final mantissa should be 24 bits with a leading 1.
+ // We shift it so that it occupies 25 bits with a leading 1.
+ ///////
+ uint64_t upperbit = upper >> 63;
+ uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+ lz += int(1 ^ upperbit);
+
+ // Here we have mantissa < (1<<25).
+ int64_t real_exponent = exponent - lz;
+ if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+ // Here we have that real_exponent <= 0 so -real_exponent >= 0
+ if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+ d = negative ? -0.0f : 0.0f;
+ return true;
+ }
+ // next line is safe because -real_exponent + 1 < 64
+ mantissa >>= -real_exponent + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0.
+ mantissa += (mantissa & 1); // round up
+ mantissa >>= 1;
+ // As in compute_float_64, rounding up may take us out of the subnormal
+ // range, so we can only decide after rounding.
+ real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+ }
+ // We have to round to even. The "to even" part is only a problem when we are
+ // right in between two floats, which we guard against. The bounds on the
+ // power of ten are those of fast_float's
+ // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+ // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+ // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+ if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+ if((mantissa << (upperbit + 38)) == upper) {
+ mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ mantissa += mantissa & 1;
+ mantissa >>= 1;
+
+ // Here we have mantissa < (1<<24), unless there was an overflow
+ if (mantissa >= (uint64_t(1) << 24)) {
+ mantissa = (uint64_t(1) << 23);
+ real_exponent++;
+ }
+ mantissa &= ~(uint64_t(1) << 23);
+ // we have to check that real_exponent is in range, otherwise we bail out
+ if (simdjson_unlikely(real_exponent > 254)) {
+ // We have an infinite value!!! We could actually throw an error here if we could.
+ return false;
+ }
+ d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+ return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<double>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+ if (atomparsing::is_valid_inf_atom(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ d = negative ? -inf : inf;
+ return true;
+ }
+
+ if (atomparsing::is_valid_nan_atom(src)) {
+ d = std::numeric_limits<float>::quiet_NaN();
+ return true;
+ }
+
+ return false;
+}
+#endif
+
// We call a fallback floating-point parser that might be slow. Note
// it will accept JSON numbers, but the JSON spec. is more restrictive so
// before you call parse_float_fallback, you need to have validated the input
@@ -37015,6 +43683,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
}
+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+ *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+ // We do not accept infinite values. See the binary64 version above for why we
+ // do not use std::isfinite.
+ return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
// check quickly whether the next 8 chars are made of digits
// at a glance, it looks better than Mula's
// http://0x80.pl/articles/swar-digits-validate.html
@@ -37033,6 +43711,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
0x3333333333333333);
}
+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+ uint32_t val;
+ // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+ static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+ std::memcpy(&val, chars, 4);
+ return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+ uint32_t val;
+ std::memcpy(&val, chars, 4);
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
template<typename I>
SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -37049,26 +43749,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
return static_cast<uint8_t>(c - '0') <= 9;
}
-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
- // we continue with the fiction that we have an integer. If the
- // floating point number is representable as x * 10^z for some integer
- // z that fits in 53 bits, then we will be able to convert back the
- // the integer into a float in a lossless manner.
- const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+ while (is_made_of_eight_digits_fast(p)) {
+ i = i * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ }
+ // A 4 to 7 digit remainder is the common case once the blocks of eight are
+ // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+ if (is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+ while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+ bool &leading_zero) {
+ if (!parse_digit(*p, i)) { leading_zero = true; return; }
+ p++;
+ leading_zero = (i == 0);
+ while (parse_digit(*p, i)) { p++; }
+}
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
#ifdef SIMDJSON_SWAR_NUMBER_PARSING
#if SIMDJSON_SWAR_NUMBER_PARSING
- // this helps if we have lots of decimals!
- // this turns out to be frequent enough.
+ // Identifiers, timestamps and counters often have eight digits or more.
+ // Prior related work: jsonifier parses integers as eight-digit SWAR words
+ // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+ // takes one such word with parse_eight_digits_unrolled, the routine
+ // simdjson uses for long fractions.
if (is_made_of_eight_digits_fast(p)) {
i = i * 100000000 + parse_eight_digits_unrolled(p);
p += 8;
}
+ const uint8_t *const swar_end = p + 8;
+ while (p < swar_end && is_made_of_four_digits_fast(p)) {
+ i = i * 10000 + parse_four_digits_unrolled(p);
+ p += 4;
+ }
#endif // SIMDJSON_SWAR_NUMBER_PARSING
#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
- // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
- if (parse_digit(*p, i)) { ++p; }
while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+ // we continue with the fiction that we have an integer. If the
+ // floating point number is representable as x * 10^z for some integer
+ // z that fits in 53 bits, then we will be able to convert back the
+ // the integer into a float in a lossless manner.
+ const uint8_t *const first_after_period = p;
+ parse_fraction_digits(p, i);
exponent = first_after_period - p;
// Decimal without digits (123.) is illegal
if (exponent == 0) {
@@ -37157,7 +43904,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
} // unnamed namespace
/** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
if (parse_float_fallback(src, answer)) {
return SUCCESS;
}
@@ -37240,15 +43987,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
return SUCCESS; // always succeeds
}
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
#else
// parse the number at src
@@ -37279,7 +44028,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
size_t digit_count = size_t(p - start_digits);
- if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+ if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // By this point, we know that our input does not begin with a digit. We will attempt
+ // to handle NaN/Infinity.
+
+ double d;
+ if (compute_nan_inf(p, negative, d)) {
+ writer.append_double(d);
+ return SUCCESS;
+ }
+#endif
+
+ return INVALID_NUMBER(src);
+ }
//
// Handle floats if there is a . or e (or both)
@@ -37328,7 +44091,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
// - Therefore, if the number is positive and lower than that, it's overflow.
// - The value we are looking at is less than or equal to INT64_MAX.
//
- } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+ } else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
}
// Write unsigned if it does not fit in a signed integer.
@@ -37427,7 +44190,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37525,7 +44288,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37580,7 +44343,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = p;
uint64_t i = 0;
- while (parse_digit(*p, i)) { p++; }
+ parse_integer_digits(p, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37666,7 +44429,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
// PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
const uint8_t *const start_digits = src;
uint64_t i = 0;
- while (parse_digit(*src, i)) { src++; }
+ parse_integer_digits(src, i);
// If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
// Optimization note: size_t is expected to be unsigned.
@@ -37706,11 +44469,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
+ if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no loading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ double d;
+ if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+ return INCORRECT_TYPE;
+ }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -37721,9 +44493,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37772,67 +44543,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
return d;
}
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
- return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
- return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
- bool negative = (*src == '-');
- src += uint8_t(negative);
- const uint8_t *p = src;
- while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
- size_t digit_count = size_t(p - src);
- if ( p == src ) { return NUMBER_ERROR; }
- if (jsoncharutils::is_structural_or_whitespace(*p)) {
- static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
- // We have an integer.
- if(simdjson_unlikely(digit_count > 20)) {
- return number_type::big_integer;
- }
- // If the number is negative and valid, it must be a signed integer.
- if(negative) {
- if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
- if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
- return number_type::big_integer;
- }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
- if(digit_count == 1 && src[0] == '0') {
- // We have to write -0.0 instead of 0
- return number_type::floating_point_number;
- }
-#endif
- return number_type::signed_integer;
- }
- // Let us check if we have a big integer (>=2**64).
- static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
- if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
- return number_type::big_integer;
- }
- // The number is positive and smaller than 18446744073709551616 (or 2**64).
- // We want values larger or equal to 9223372036854775808 to be unsigned
- // integers, and the other values to be signed integers.
- if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
- return number_type::unsigned_integer;
- }
- return number_type::signed_integer;
- }
- // Hopefully, we have 'e' or 'E' or '.'.
- return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
- if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
//
// Check for minus sign
//
@@ -37844,91 +44560,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
//
uint64_t i = 0;
const uint8_t *p = src;
- if(p == src_end) { return NUMBER_ERROR; }
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
// no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
- if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+ if ( p == src ) {
- //
- // Parse the decimal part.
- //
- int64_t exponent = 0;
- bool overflow;
- if (simdjson_likely((p != src_end) && (*p == '.'))) {
- p++;
- const uint8_t *start_decimal_digits = p;
- if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while ((p != src_end) && parse_digit(*p, i)) { p++; }
- exponent = -(p - start_decimal_digits);
-
- // Overflow check. More than 19 digits (minus the decimal) may be overflow.
- overflow = p-src-1 > 19;
- if (simdjson_unlikely(overflow && leading_zero)) {
- // Skip leading 0.00000 and see if it still overflows
- const uint8_t *start_digits = src + 2;
- while (*start_digits == '0') { start_digits++; }
- overflow = start_digits-src > 19;
- }
- } else {
- overflow = p-src > 19;
- }
-
- //
- // Parse the exponent
- //
- if ((p != src_end) && (*p == 'e' || *p == 'E')) {
- p++;
- if(p == src_end) { return NUMBER_ERROR; }
- bool exp_neg = *p == '-';
- p += exp_neg || *p == '+';
-
- uint64_t exp = 0;
- const uint8_t *start_exp_digits = p;
- while ((p != src_end) && parse_digit(*p, exp)) { p++; }
- // no exp digits, or 20+ exp digits
- if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
- exponent += exp_neg ? 0-exp : exp;
- }
-
- if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
- overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, the number may be nan or infinity.
+ // Attempt to compute those, and return on success.
+ float f;
+ if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
- //
- // Assemble (or slow-parse) the float
- //
- double d;
- if (simdjson_likely(!overflow)) {
- if (compute_float_64(exponent, i, negative, d)) { return d; }
- }
- if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
- return NUMBER_ERROR;
+ return INCORRECT_TYPE;
}
- return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
- //
- // Check for minus sign
- //
- bool negative = (*(src + 1) == '-');
- src += uint8_t(negative) + 1;
-
- //
- // Parse the integer part.
- //
- uint64_t i = 0;
- const uint8_t *p = src;
- p += parse_digit(*p, i);
- bool leading_zero = (i == 0);
- while (parse_digit(*p, i)) { p++; }
- // no integer digits, or 0123 (zero must be solo)
- if ( p == src ) { return INCORRECT_TYPE; }
if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
//
@@ -37939,9 +44584,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
if (simdjson_likely(*p == '.')) {
p++;
const uint8_t *start_decimal_digits = p;
- if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
- p++;
- while (parse_digit(*p, i)) { p++; }
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+ return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+ return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+ const uint8_t *p = src;
+ while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+ size_t digit_count = size_t(p - src);
+ if ( p == src ) { return NUMBER_ERROR; }
+ if (jsoncharutils::is_structural_or_whitespace(*p)) {
+ static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+ // We have an integer.
+ if(simdjson_unlikely(digit_count > 20)) {
+ return number_type::big_integer;
+ }
+ // If the number is negative and valid, it must be a signed integer.
+ if(negative) {
+ if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+ if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+ return number_type::big_integer;
+ }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+ if(digit_count == 1 && src[0] == '0') {
+ // We have to write -0.0 instead of 0
+ return number_type::floating_point_number;
+ }
+#endif
+ return number_type::signed_integer;
+ }
+ // Let us check if we have a big integer (>=2**64).
+ static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+ if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+ return number_type::big_integer;
+ }
+ // The number is positive and smaller than 18446744073709551616 (or 2**64).
+ // We want values larger or equal to 9223372036854775808 to be unsigned
+ // integers, and the other values to be signed integers.
+ if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+ return number_type::unsigned_integer;
+ }
+ return number_type::signed_integer;
+ }
+ // Hopefully, we have 'e' or 'E' or '.'.
+ return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+ if(src == src_end) { return NUMBER_ERROR; }
+ //
+ // Check for minus sign
+ //
+ bool negative = (*src == '-');
+ src += uint8_t(negative);
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ if(p == src_end) { return NUMBER_ERROR; }
+ p += parse_digit(*p, i);
+ bool leading_zero = (i == 0);
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) { return INCORRECT_TYPE; }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely((p != src_end) && (*p == '.'))) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+ p++;
+ while ((p != src_end) && parse_digit(*p, i)) { p++; }
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = start_digits-src > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+ p++;
+ if(p == src_end) { return NUMBER_ERROR; }
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ double d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_64(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ double inf = std::numeric_limits<double>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<double>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
exponent = -(p - start_decimal_digits);
// Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37990,6 +44865,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
return d;
}
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+ //
+ // Check for minus sign
+ //
+ bool negative = (*(src + 1) == '-');
+ src += uint8_t(negative) + 1;
+
+ //
+ // Parse the integer part.
+ //
+ uint64_t i = 0;
+ const uint8_t *p = src;
+ bool leading_zero;
+ parse_float_integer_digits(p, i, leading_zero);
+ // no integer digits, or 0123 (zero must be solo)
+ if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // If there are no leading digits, attempt to parse numbers that are either
+ // NaN or Infinity
+ if (atomparsing::is_valid_inf_in_string(src)) {
+ float inf = std::numeric_limits<float>::infinity();
+ return negative ? -inf : inf;
+ }
+
+ if (atomparsing::is_valid_nan_in_string(src)) {
+ return std::numeric_limits<float>::quiet_NaN();
+ }
+#endif
+
+ return INCORRECT_TYPE;
+ }
+ if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+ //
+ // Parse the decimal part.
+ //
+ int64_t exponent = 0;
+ bool overflow;
+ if (simdjson_likely(*p == '.')) {
+ p++;
+ const uint8_t *start_decimal_digits = p;
+ parse_fraction_digits(p, i);
+ if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+ exponent = -(p - start_decimal_digits);
+
+ // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+ overflow = p-src-1 > 19;
+ if (simdjson_unlikely(overflow && leading_zero)) {
+ // Skip leading 0.00000 and see if it still overflows
+ const uint8_t *start_digits = src + 2;
+ while (*start_digits == '0') { start_digits++; }
+ overflow = p-start_digits > 19;
+ }
+ } else {
+ overflow = p-src > 19;
+ }
+
+ //
+ // Parse the exponent
+ //
+ if (*p == 'e' || *p == 'E') {
+ p++;
+ bool exp_neg = *p == '-';
+ p += exp_neg || *p == '+';
+
+ uint64_t exp = 0;
+ const uint8_t *start_exp_digits = p;
+ while (parse_digit(*p, exp)) { p++; }
+ // no exp digits, or 20+ exp digits
+ if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+ exponent += exp_neg ? 0-exp : exp;
+ }
+
+ if (*p != '"') { return NUMBER_ERROR; }
+
+ overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+ //
+ // Assemble (or slow-parse) the float
+ //
+ float d;
+ if (simdjson_likely(!overflow)) {
+ if (compute_float_32(exponent, i, negative, d)) { return d; }
+ }
+ if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+ return NUMBER_ERROR;
+ }
+ return d;
+}
+
} // unnamed namespace
#endif // SIMDJSON_SKIPNUMBERPARSING
@@ -38172,6 +45140,376 @@ simdjson_inline implementation_simdjson_result_base<T>::implementation_simdjson_
// Otherwise, amalgamation will fail.
/* skipped duplicate #include "simdjson/concepts.h" */
/* skipped duplicate #include "simdjson/dom/fractured_json.h" */
+/* including simdjson/annotations.h: #include "simdjson/annotations.h" */
+/* begin file simdjson/annotations.h */
+#ifndef SIMDJSON_ANNOTATIONS_H
+#define SIMDJSON_ANNOTATIONS_H
+
+/**
+ * @file annotations.h
+ * @brief Provides compile-time annotations for simdjson structures.
+ * This header defines annotations that can be applied to data members of structures
+ * (and to the structures and enumerations themselves) to control how they are
+ * serialized/deserialized with simdjson. The set of annotations is modelled after
+ * the attributes of the Rust serde library.
+ *
+ * Member annotations:
+ *
+ * [[= simdjson::rename<"name">]] use "name" as the JSON key
+ * [[= simdjson::alias<"a", "b">]] also accept "a" and "b" when deserializing
+ * [[= simdjson::skip]] never serialize nor deserialize
+ * [[= simdjson::skip_serializing]] never serialize
+ * [[= simdjson::skip_deserializing]] never deserialize (keeps its current value)
+ * [[= simdjson::skip_serializing_if<pred>]] do not serialize when pred(value) is true
+ * [[= simdjson::default_value]] a missing key is not an error
+ * [[= simdjson::default_from<factory>]] a missing key sets the member to factory()
+ * [[= simdjson::with<Adapter>]] custom (de)serialization via Adapter
+ * [[= simdjson::flatten]] inline the members of a nested structure
+ *
+ * Structure (container) annotations:
+ *
+ * [[= simdjson::rename_all<simdjson::case_style::camel_case>]] rename every member
+ * [[= simdjson::default_value]] no missing key is an error
+ * [[= simdjson::deny_unknown_fields]] unknown keys are a deserialization error
+ * [[= simdjson::transparent]] (de)serialize as the single member
+ *
+ * Enumeration annotations: rename_all on the enumeration, rename and alias on the
+ * enumerators (e.g., `enum class color { red [[= simdjson::rename<"RED">]] };`).
+ *
+ * This is currently experimental and subject to change (syntax and semantics may evolve).
+ */
+
+#if SIMDJSON_STATIC_REFLECTION
+
+#include <meta>
+#include <string>
+#include <string_view>
+#include <vector>
+
+namespace simdjson {
+
+// Structural compile-time string -- char array avoids the pointer-based
+// 'reflect_constant failed' that occurs with const char* / string_view members.
+template <size_t N>
+struct fixed_string {
+ char data[N];
+
+ consteval fixed_string(const char (&s)[N]) noexcept {
+ for (size_t i = 0; i < N; ++i) { data[i] = s[i]; }
+ }
+
+ consteval std::string_view view() const noexcept { return {data, N - 1}; }
+
+ consteval bool operator==(const fixed_string&) const noexcept = default;
+};
+
+/**
+ * Naming conventions for simdjson::rename_all, mirroring serde's rename_all.
+ * Except for lowercase and uppercase, the C++ identifier is first split into
+ * words at underscores and at case changes ("userId", "user_id" and "UserId" all
+ * give the words "user" and "id"; "HTTPServer" gives "HTTP" and "Server"), and
+ * the words are then joined according to the convention.
+ */
+enum class case_style {
+ lowercase, ///< every letter lowercased, nothing else changes: userId -> userid
+ uppercase, ///< every letter uppercased, nothing else changes: user_id -> USER_ID
+ pascal_case, ///< UserId
+ camel_case, ///< userId
+ snake_case, ///< user_id
+ screaming_snake_case, ///< USER_ID
+ kebab_case, ///< user-id
+ screaming_kebab_case ///< USER-ID
+};
+
+namespace detail {
+ template <fixed_string Name>
+ struct rename_t {
+ static constexpr auto name = Name;
+ // Exposed as a pointer and a size: std::meta::extract requires structural types.
+ static constexpr const char *key_data = Name.data;
+ static constexpr size_t key_size = Name.view().size();
+ };
+ template <fixed_string... Names>
+ struct alias_t {
+ static_assert(sizeof...(Names) > 0, "simdjson::alias requires at least one name");
+ static constexpr std::string_view keys[] = {Names.view()...};
+ static constexpr const std::string_view *keys_data = keys;
+ static constexpr size_t keys_count = sizeof...(Names);
+ };
+ struct skip_tag {};
+ struct skip_serializing_tag {};
+ struct skip_deserializing_tag {};
+ template <auto Predicate>
+ struct skip_serializing_if_t {
+ static constexpr auto predicate = Predicate;
+ };
+ struct default_value_tag {};
+ template <auto Factory>
+ struct default_from_t {
+ static constexpr auto factory = Factory;
+ };
+ template <typename Adapter>
+ struct with_t {
+ using adapter = Adapter;
+ };
+ template <case_style Style>
+ struct rename_all_t {
+ static constexpr case_style style = Style;
+ };
+ struct deny_unknown_fields_tag {};
+ struct transparent_tag {};
+ struct flatten_tag {};
+
+ // Predicates usable with skip_serializing_if.
+ struct is_none_t {
+ template <typename T>
+ constexpr bool operator()(const T& v) const noexcept { return !v; }
+ };
+ struct is_empty_t {
+ template <typename T>
+ constexpr bool operator()(const T& v) const noexcept { return v.empty(); }
+ };
+} // namespace detail
+
+// Usage: [[= simdjson::rename<"first_name">]] std::string firstName;
+template <fixed_string Name>
+inline constexpr detail::rename_t<Name> rename{};
+
+// Usage: [[= simdjson::alias<"userName", "login">]] std::string user_name;
+// The aliases are accepted (in addition to the regular key) when deserializing.
+// Serialization always uses the regular key. If the JSON object contains more
+// than one of the names, which one is used is unspecified.
+template <fixed_string... Names>
+inline constexpr detail::alias_t<Names...> alias{};
+
+// Usage: [[= simdjson::skip]] int internalCache;
+inline constexpr detail::skip_tag skip{};
+
+// Usage: [[= simdjson::skip_serializing]] std::string password;
+inline constexpr detail::skip_serializing_tag skip_serializing{};
+
+// Usage: [[= simdjson::skip_deserializing]] int computed;
+// The member is never assigned during deserialization (it keeps its current
+// value) and a matching key in the JSON input is treated as unknown.
+inline constexpr detail::skip_deserializing_tag skip_deserializing{};
+
+// Usage: [[= simdjson::skip_serializing_if<simdjson::is_none>]] std::optional<int> x;
+// The predicate is called with the member value; when it returns true, the key
+// is omitted from the output.
+template <auto Predicate>
+inline constexpr detail::skip_serializing_if_t<Predicate> skip_serializing_if{};
+
+// Predicate: true for an empty std::optional, a null smart pointer, etc.
+inline constexpr detail::is_none_t is_none{};
+// Predicate: true for an empty string or container.
+inline constexpr detail::is_empty_t is_empty{};
+
+// Usage: [[= simdjson::default_value]] int port = 8080;
+// When the key is missing from the JSON input, the member is left untouched
+// (with get<T>(), it keeps its default member initializer) instead of reporting
+// NO_SUCH_FIELD. Applied to a structure, it applies to all of its members.
+inline constexpr detail::default_value_tag default_value{};
+
+// Usage: [[= simdjson::default_from<make_port>]] int port;
+// When the key is missing from the JSON input, the member is assigned the
+// result of calling the factory (a constexpr callable taking no argument, such
+// as a captureless lambda or a pointer to a function).
+template <auto Factory>
+inline constexpr detail::default_from_t<Factory> default_from{};
+
+// Usage: [[= simdjson::with<unix_time>]] std::chrono::system_clock::time_point t;
+// Adapter is a type that provides one or both of
+// static void serialize(simdjson::builder::string_builder &b, const T &value);
+// static simdjson::error_code deserialize(simdjson::ondemand::value &v, T &out);
+// (the parameters may also be declared auto&). When one of them is missing, the
+// default behaviour is used in that direction.
+template <typename Adapter>
+inline constexpr detail::with_t<Adapter> with{};
+
+// Usage: [[= simdjson::flatten]] pagination page;
+// The members of the nested structure are (de)serialized as if they were members
+// of the enclosing structure: {"id":1,"limit":10,"offset":0} rather than
+// {"id":1,"page":{"limit":10,"offset":0}}. The nested structure's own annotations
+// (rename_all, default_value, ...) apply to its members.
+inline constexpr detail::flatten_tag flatten{};
+
+// Usage: struct [[= simdjson::rename_all<simdjson::case_style::camel_case>]] S {...};
+// Also applies to enumerations. An explicit rename on a member takes precedence.
+template <case_style Style>
+inline constexpr detail::rename_all_t<Style> rename_all{};
+
+// Usage: struct [[= simdjson::deny_unknown_fields]] S {...};
+// Deserialization fails with UNKNOWN_FIELD when the JSON object has a key that
+// is not deserialized into a member (including the keys of skipped members).
+inline constexpr detail::deny_unknown_fields_tag deny_unknown_fields{};
+
+// Usage: struct [[= simdjson::transparent]] user_id { int64_t value; };
+// A structure with a single data member is (de)serialized as that member alone:
+// user_id{42} becomes 42 rather than {"value":42}.
+inline constexpr detail::transparent_tag transparent{};
+
+namespace detail {
+
+// True when the entity (a data member, an enumerator or a type) carries an
+// annotation of type tag.
+consteval bool has_annotation(std::meta::info entity, std::meta::info tag) {
+ return !std::meta::annotations_of_with_type(entity, tag).empty();
+}
+
+// Returns the type of the (first) annotation of entity that is a specialization
+// of the class template tmpl, or std::meta::info{} when there is none.
+consteval std::meta::info annotation_of_template(std::meta::info entity, std::meta::info tmpl) {
+ for (std::meta::info ann : std::meta::annotations_of(entity)) {
+ std::meta::info type = std::meta::type_of(ann);
+ if (std::meta::has_template_arguments(type) && std::meta::template_of(type) == tmpl) {
+ return type;
+ }
+ }
+ return std::meta::info{};
+}
+
+// Value of the static data member `name` of the class `type`.
+template <typename T>
+consteval T static_member_value(std::meta::info type, std::string_view name) {
+ for (std::meta::info m : std::meta::static_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == name) {
+ return std::meta::extract<T>(m);
+ }
+ }
+ return T{};
+}
+
+consteval bool is_upper(char c) { return c >= 'A' && c <= 'Z'; }
+consteval bool is_lower(char c) { return c >= 'a' && c <= 'z'; }
+consteval bool is_digit(char c) { return c >= '0' && c <= '9'; }
+consteval char to_upper(char c) { return is_lower(c) ? char(c - 'a' + 'A') : c; }
+consteval char to_lower(char c) { return is_upper(c) ? char(c - 'A' + 'a') : c; }
+
+// Split an identifier into words: at underscores, at a lowercase letter or digit
+// followed by an uppercase letter (userId), and before the last capital of an
+// acronym followed by a lowercase letter (HTTPServer -> HTTP, Server).
+consteval std::vector<std::string> split_identifier(std::string_view id) {
+ std::vector<std::string> words;
+ std::string current;
+ for (size_t i = 0; i < id.size(); i++) {
+ char c = id[i];
+ if (c == '_') {
+ if (!current.empty()) { words.push_back(current); current.clear(); }
+ continue;
+ }
+ if (is_upper(c) && !current.empty()) {
+ char prev = current.back();
+ bool next_is_lower = (i + 1 < id.size()) && is_lower(id[i + 1]);
+ if (is_lower(prev) || is_digit(prev) || (is_upper(prev) && next_is_lower)) {
+ words.push_back(current);
+ current.clear();
+ }
+ }
+ current.push_back(c);
+ }
+ if (!current.empty()) { words.push_back(current); }
+ return words;
+}
+
+consteval std::string apply_case_style(std::string_view id, case_style style) {
+ std::string result;
+ if (style == case_style::lowercase || style == case_style::uppercase) {
+ for (char c : id) {
+ result.push_back(style == case_style::lowercase ? to_lower(c) : to_upper(c));
+ }
+ return result;
+ }
+ std::vector<std::string> words = split_identifier(id);
+ for (size_t w = 0; w < words.size(); w++) {
+ const std::string &word = words[w];
+ if (style == case_style::pascal_case || style == case_style::camel_case) {
+ for (size_t i = 0; i < word.size(); i++) {
+ bool capital = (i == 0) && (style == case_style::pascal_case || w > 0);
+ result.push_back(capital ? to_upper(word[i]) : to_lower(word[i]));
+ }
+ } else {
+ bool upper = style == case_style::screaming_snake_case || style == case_style::screaming_kebab_case;
+ bool kebab = style == case_style::kebab_case || style == case_style::screaming_kebab_case;
+ if (w > 0) { result.push_back(kebab ? '-' : '_'); }
+ for (char c : word) { result.push_back(upper ? to_upper(c) : to_lower(c)); }
+ }
+ }
+ return result;
+}
+
+// The JSON key for a data member or an enumerator: an explicit rename wins,
+// then the rename_all of the enclosing structure or enumeration, then the C++
+// identifier.
+consteval std::string_view json_key_name(std::meta::info entity) {
+ std::meta::info rename_type = annotation_of_template(entity, ^^rename_t);
+ if (rename_type != std::meta::info{}) {
+ return std::define_static_string(std::string_view{
+ static_member_value<const char *>(rename_type, "key_data"),
+ static_member_value<size_t>(rename_type, "key_size")});
+ }
+ std::meta::info rename_all_type = annotation_of_template(std::meta::parent_of(entity), ^^rename_all_t);
+ if (rename_all_type != std::meta::info{}) {
+ case_style style = static_member_value<case_style>(rename_all_type, "style");
+ return std::define_static_string(apply_case_style(std::meta::identifier_of(entity), style));
+ }
+ return std::define_static_string(std::meta::identifier_of(entity));
+}
+
+// The keys accepted when deserializing a data member or an enumerator: the JSON
+// key first, followed by the aliases (if any), in declaration order.
+consteval std::vector<std::string_view> json_key_names(std::meta::info entity) {
+ std::vector<std::string_view> names{json_key_name(entity)};
+ for (std::meta::info ann : std::meta::annotations_of(entity)) {
+ std::meta::info type = std::meta::type_of(ann);
+ if (std::meta::has_template_arguments(type) && std::meta::template_of(type) == ^^alias_t) {
+ const std::string_view *keys = static_member_value<const std::string_view *>(type, "keys_data");
+ size_t count = static_member_value<size_t>(type, "keys_count");
+ for (size_t i = 0; i < count; i++) { names.push_back(keys[i]); }
+ }
+ }
+ return names;
+}
+
+// The structure type of a member annotated with flatten.
+consteval std::meta::info flattened_type(std::meta::info mem) {
+ if (std::meta::is_reference_type(std::meta::type_of(mem))) {
+ // A reference member could refer back to the enclosing structure: the
+ // flattening would never terminate.
+ throw std::meta::exception(u8"simdjson::flatten requires a member that is not a reference", mem);
+ }
+ std::meta::info type = std::meta::remove_cvref(std::meta::type_of(mem));
+ if (!std::meta::is_class_type(type)) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member of class type", mem);
+ }
+ return type;
+}
+
+// The single data member of a structure annotated with transparent: the only
+// member that is not annotated with skip.
+consteval std::meta::info transparent_member(std::meta::info type) {
+ std::vector<std::meta::info> members;
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!has_annotation(mem, ^^skip_tag)) { members.push_back(mem); }
+ }
+ if (members.size() != 1) {
+ throw std::meta::exception(u8"simdjson::transparent requires exactly one data member (not counting skipped members)", type);
+ }
+ return members[0];
+}
+
+} // namespace detail
+
+// Returns the JSON key for a reflected data member (or enumerator).
+template <auto dm>
+consteval const char* get_json_key_name() {
+ return detail::json_key_name(dm).data();
+}
+
+} // namespace simdjson
+
+#endif // SIMDJSON_STATIC_REFLECTION
+#endif // SIMDJSON_ANNOTATIONS_H
+/* end file simdjson/annotations.h */
#endif // SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H
/* end file simdjson/generic/builder/dependencies.h */
@@ -38297,7 +45635,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {
/* result might be undefined when input_num is zero */
simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+ // if the system supports SVE or CSSC, __builtin_popcountll
+ // might be compiled to fewer single instructions. For CSSC,
+ // __builtin_popcountll is compiled to a single instruction.
+ return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
}
@@ -38334,15 +45679,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace arm64
@@ -38606,6 +45942,7 @@ namespace {
return vget_lane_u64(
vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
}
+ // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
};
@@ -39240,7 +46577,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -39254,6 +46591,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -39305,6 +46651,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -39318,7 +46684,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -39375,6 +46741,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -39392,64 +46759,370 @@ namespace simdjson {
namespace arm64 {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -39459,92 +47132,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -39552,20 +47392,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -39573,14 +47415,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -39594,39 +47436,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -39637,22 +47455,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -39665,40 +47501,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -39711,25 +47548,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace arm64
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = arm64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- arm64::builder::string_builder b(initial_capacity);
- arm64::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return arm64::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = arm64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- arm64::builder::string_builder b(initial_capacity);
- arm64::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return arm64::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -39875,6 +47705,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for arm64: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for arm64 */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -39909,6 +47740,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -39928,6 +47764,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -39935,6 +47774,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -39984,105 +47826,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -40173,6 +47916,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -40328,6 +48116,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -40357,9 +48393,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -40382,7 +48422,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -40438,81 +48478,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -40540,87 +48635,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -40681,7 +48751,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -40700,6 +48772,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -40729,7 +48809,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -41335,7 +49415,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -41349,6 +49429,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -41400,6 +49489,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -41413,7 +49522,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -41470,6 +49579,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -41487,64 +49597,370 @@ namespace simdjson {
namespace fallback {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -41554,92 +49970,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -41647,20 +50230,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -41668,14 +50253,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -41689,39 +50274,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -41732,22 +50293,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -41760,40 +50339,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -41806,25 +50386,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace fallback
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = fallback::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- fallback::builder::string_builder b(initial_capacity);
- fallback::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return fallback::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = fallback::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- fallback::builder::string_builder b(initial_capacity);
- fallback::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return fallback::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -41970,6 +50543,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for fallback: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for fallback */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -42004,6 +50578,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -42023,6 +50602,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -42030,6 +50612,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -42079,105 +50664,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -42268,6 +50754,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -42423,6 +50954,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -42452,9 +51231,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -42477,7 +51260,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -42533,81 +51316,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -42635,87 +51473,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -42776,7 +51589,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -42795,6 +51610,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -42824,7 +51647,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -43169,16 +51992,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace haswell
@@ -43917,7 +52730,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -43931,6 +52744,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -43982,6 +52804,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -43995,7 +52837,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -44052,6 +52894,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -44069,64 +52912,370 @@ namespace simdjson {
namespace haswell {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -44136,92 +53285,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
}
- b.append(']');
}
-// append functions that delegate to atom functions for primitive types
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
+ }
+ writer w(b);
+ atom(w, t);
+ w.sync();
+}
+
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -44229,20 +53545,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -44250,14 +53568,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -44271,39 +53589,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -44314,22 +53608,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -44342,40 +53654,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -44388,25 +53701,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace haswell
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = haswell::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- haswell::builder::string_builder b(initial_capacity);
- haswell::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return haswell::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = haswell::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- haswell::builder::string_builder b(initial_capacity);
- haswell::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return haswell::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -44552,6 +53858,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for haswell: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for haswell */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -44586,6 +53893,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -44605,6 +53917,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -44612,6 +53927,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -44661,105 +53979,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -44850,6 +54069,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -45005,6 +54269,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -45034,9 +54546,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -45059,7 +54575,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -45115,81 +54631,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -45217,87 +54788,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -45358,7 +54904,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -45377,6 +54925,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -45406,7 +54962,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -45748,16 +55304,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace icelake
@@ -46499,7 +56045,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -46513,6 +56059,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -46564,6 +56119,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -46577,7 +56152,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -46634,6 +56209,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -46651,64 +56227,370 @@ namespace simdjson {
namespace icelake {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -46718,92 +56600,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
}
- b.append(']');
}
-// append functions that delegate to atom functions for primitive types
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
+ }
+ writer w(b);
+ atom(w, t);
+ w.sync();
+}
+
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -46811,20 +56860,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -46832,14 +56883,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -46853,39 +56904,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -46896,22 +56923,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -46924,40 +56969,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -46970,25 +57016,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace icelake
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = icelake::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- icelake::builder::string_builder b(initial_capacity);
- icelake::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return icelake::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = icelake::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- icelake::builder::string_builder b(initial_capacity);
- icelake::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return icelake::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -47134,6 +57173,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for icelake: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for icelake */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -47168,6 +57208,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -47187,6 +57232,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -47194,6 +57242,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -47243,105 +57294,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -47432,6 +57384,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -47587,6 +57584,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -47616,9 +57861,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -47641,7 +57890,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -47697,81 +57946,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -47799,87 +58103,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -47940,7 +58219,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -47959,6 +58240,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -47988,7 +58277,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -48302,16 +58591,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace ppc64
@@ -49196,7 +59475,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -49210,6 +59489,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -49261,6 +59549,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -49274,7 +59582,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -49331,6 +59639,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -49348,64 +59657,370 @@ namespace simdjson {
namespace ppc64 {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -49415,92 +60030,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -49508,20 +60290,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -49529,14 +60313,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -49550,39 +60334,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -49593,22 +60353,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -49621,40 +60399,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -49667,25 +60446,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace ppc64
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = ppc64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- ppc64::builder::string_builder b(initial_capacity);
- ppc64::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return ppc64::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = ppc64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- ppc64::builder::string_builder b(initial_capacity);
- ppc64::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return ppc64::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -49831,6 +60603,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for ppc64: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for ppc64 */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -49865,6 +60638,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -49884,6 +60662,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -49891,6 +60672,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -49940,105 +60724,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -50129,6 +60814,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -50284,6 +61014,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -50313,9 +61291,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -50338,7 +61320,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -50394,81 +61376,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -50496,87 +61533,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -50637,7 +61649,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -50656,6 +61670,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -50685,7 +61707,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -51013,16 +62035,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -51601,16 +62613,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -52210,7 +63212,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -52224,6 +63226,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -52275,6 +63286,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -52288,7 +63319,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -52345,6 +63376,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -52362,64 +63394,370 @@ namespace simdjson {
namespace westmere {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -52429,92 +63767,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -52522,20 +64027,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -52543,14 +64050,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -52564,39 +64071,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -52607,22 +64090,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -52635,40 +64136,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -52681,25 +64183,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace westmere
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = westmere::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- westmere::builder::string_builder b(initial_capacity);
- westmere::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return westmere::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = westmere::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- westmere::builder::string_builder b(initial_capacity);
- westmere::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return westmere::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -52845,6 +64340,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for westmere: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for westmere */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -52879,6 +64375,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -52898,6 +64399,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -52905,6 +64409,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -52954,105 +64461,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -53143,6 +64551,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -53298,6 +64751,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -53327,9 +65028,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -53352,7 +65057,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -53408,81 +65113,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -53510,87 +65270,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -53651,7 +65386,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -53670,6 +65407,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -53699,7 +65444,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -53980,10 +65725,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lsx
@@ -54698,7 +66439,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -54712,6 +66453,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -54763,6 +66513,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -54776,7 +66546,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -54833,6 +66603,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -54850,64 +66621,370 @@ namespace simdjson {
namespace lsx {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -54917,92 +66994,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -55010,20 +67254,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -55031,14 +67277,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -55052,39 +67298,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -55095,22 +67317,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -55123,40 +67363,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -55169,25 +67410,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace lsx
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = lsx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- lsx::builder::string_builder b(initial_capacity);
- lsx::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return lsx::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = lsx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- lsx::builder::string_builder b(initial_capacity);
- lsx::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return lsx::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -55333,6 +67567,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for lsx: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for lsx */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -55367,6 +67602,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -55386,6 +67626,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -55393,6 +67636,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -55442,105 +67688,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -55631,6 +67778,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -55786,6 +67978,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -55815,9 +68255,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -55840,7 +68284,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -55896,81 +68340,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -55998,87 +68497,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -56139,7 +68613,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -56158,6 +68634,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -56187,7 +68671,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -56473,10 +68957,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lasx
@@ -57209,7 +69689,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -57223,6 +69703,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -57274,6 +69763,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -57287,7 +69796,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -57344,6 +69853,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -57361,64 +69871,370 @@ namespace simdjson {
namespace lasx {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -57428,92 +70244,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -57521,20 +70504,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -57542,14 +70527,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -57563,39 +70548,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -57606,22 +70567,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -57634,40 +70613,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -57680,25 +70660,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace lasx
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = lasx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- lasx::builder::string_builder b(initial_capacity);
- lasx::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return lasx::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = lasx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- lasx::builder::string_builder b(initial_capacity);
- lasx::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return lasx::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -57844,6 +70817,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for lasx: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for lasx */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -57878,6 +70852,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -57897,6 +70876,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -57904,6 +70886,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -57953,105 +70938,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -58142,6 +71028,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -58297,6 +71228,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -58326,9 +71505,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -58351,7 +71534,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -58407,81 +71590,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -58509,87 +71747,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -58650,7 +71863,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -58669,6 +71884,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -58698,7 +71921,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -58993,11 +72216,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
return __builtin_popcountll(input_num);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace rvv_vls
@@ -59724,7 +72942,7 @@ public:
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void append(const R &range) noexcept;
#endif
/**
@@ -59738,6 +72956,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
* There is no UTF-8 validation.
*/
simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+ /**
+ * Append exactly N characters from str. The length is a template parameter
+ * so the compiler can fully inline the memcpy with a compile-time-constant
+ * size, avoiding the libc call. Used for compile-time-constant keys in the
+ * reflection struct atom.
+ */
+ template <size_t N>
+ simdjson_inline void append_raw_n(const char *str) noexcept;
#if SIMDJSON_EXCEPTIONS
/**
* Creates an std::string from the written JSON buffer.
@@ -59789,6 +73016,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
*/
simdjson_inline size_t size() const noexcept;
+ // ============================================================
+ // Internal hooks for the position-as-local writer in json_builder.h.
+ // These exist so the reflection atom code can hold buffer pointer,
+ // position and capacity in registers across long write chains rather
+ // than reloading them after every char* write (strict aliasing
+ // forces those reloads when accessed via members of *this). User
+ // code should NOT call these directly.
+ // ============================================================
+ simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+ simdjson_inline size_t unsafe_position() const noexcept { return position; }
+ simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+ simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+ /// Make capacity available for at least `n` more bytes after the current
+ /// position. Returns false if the allocation failed.
+ simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+ grow_buffer(needed_total_capacity);
+ return is_valid;
+ }
+ simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
private:
/**
* Returns true if we can write at least upcoming_bytes bytes.
@@ -59802,7 +73049,7 @@ private:
* If the allocation fails, is_valid is set to false. We expect
* that this function would not be repeatedly called.
*/
- simdjson_inline void grow_buffer(size_t desired_capacity);
+ inline void grow_buffer(size_t desired_capacity);
/**
* We use this helper function to make sure that is_valid is kept consistent.
@@ -59859,6 +73106,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
/* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
/* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#if SIMDJSON_STATIC_REFLECTION
@@ -59876,64 +73124,370 @@ namespace simdjson {
namespace rvv_vls {
namespace builder {
-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+ static constexpr bool checked = Checked;
+ char *ptr; // buffer pointer (refreshed after a grow)
+ size_t pos; // write position (local)
+ size_t cap; // capacity (refreshed after a grow)
+ string_builder &sb; // back-ref for grow / sync
+
+ // Snapshot string_builder state into a writer for the duration of
+ // a write chain.
+ simdjson_really_inline basic_writer(string_builder &builder) noexcept
+ : ptr(builder.unsafe_data())
+ , pos(builder.unsafe_position())
+ , cap(builder.unsafe_capacity())
+ , sb(builder) {}
+
+ // Write the local position back to the underlying string_builder.
+ // Caller is responsible for invoking before the writer is dropped
+ // (otherwise data is lost). Idempotent.
+ simdjson_really_inline void sync() noexcept {
+ sb.unsafe_set_position(pos);
+ }
+
+ // Ensure at least `n` more bytes of free capacity. Grows the
+ // underlying buffer if needed (rare path). Returns false on
+ // allocation failure.
+ simdjson_really_inline bool ensure(size_t n) noexcept {
+ // pos <= cap, and cap is the size of a live allocation, so pos + n
+ // cannot wrap when n is a small constant or a compile-time length.
+ // Callers passing a size derived from input (the string atoms) must
+ // bound it against pos themselves. Keep the `pos + n <= cap` form:
+ // `n <= cap - pos` is measurably slower once the serializer is inlined.
+ if (simdjson_likely(pos + n <= cap)) { return true; }
+ return grow_slow(n);
+ }
+
+ simdjson_never_inline bool grow_slow(size_t n) noexcept {
+ // Detect overflow.
+ // This is pedantic except maybe on 32-bit targets.
+ if (simdjson_unlikely(pos + n < pos)) return false;
+ sb.unsafe_set_position(pos);
+ // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+ // so we do not need a separate overflow check here.
+ if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+ // The string_builder freed its buffer and is now invalid (null buffer,
+ // zero capacity and position). Mirror that state so that every later
+ // ensure() fails too: callers only return from the current atom, and
+ // their callers keep writing.
+ ptr = nullptr;
+ pos = 0;
+ cap = 0;
+ return false;
+ }
+ ptr = sb.unsafe_data();
+ cap = sb.unsafe_capacity();
+ return true;
+ }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+ static constexpr bool checked = false;
+ char *ptr;
+ size_t pos;
+
+ simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+ : ptr(buffer), pos(position) {}
+
+ simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+ return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+ w.sync();
+ f(w.sb);
+ w.ptr = w.sb.unsafe_data();
+ w.pos = w.sb.unsafe_position();
+ w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+ return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+ constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+ if constexpr (skip_if_type != std::meta::info{}) {
+ using skip_if = typename [: skip_if_type :];
+ return !skip_if::predicate(value);
+ } else {
+ (void)value;
+ return true;
+ }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+ call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+ } else {
+ atom(w, value);
+ }
+ } else {
+ atom(w, value);
+ }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+ // Per-field block: ensure key+value worst case, then write key + value
+ // through the writer's local pos. For arithmetic fields, the integer
+ // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+ // never round-trips through memory.
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (is_serialized_member(dm)) {
+ if (should_serialize<dm>(t.[:dm:])) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+ using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+ static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+ !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+ !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+ !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+ "simdjson::flatten requires a member whose type is a structure serialized member by member");
+ atom_fields(w, t.[:dm:], first);
+ } else {
+ // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+ // load and one store); ensure() reserves the padded length, and the
+ // unchecked writer has slack past its bound. Prior related work:
+ // jsonifier copies a power-of-two padded key and advances the cursor
+ // by the real length (serialize_impl.hpp, packed_blitter,
+ // https://github.com/nihilai-collective/Jsonifier).
+ constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+ constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+ constexpr size_t rest_key_len = first_key_len + 1;
+ constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+ constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+ std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+ if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+ w.pos += rest_key_len;
+ }
+ first = false;
+ atom_member<dm>(w, t.[:dm:]);
+ }
+ }
+ }
+ };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
auto it = t.begin();
auto end = t.end();
if (it == end) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
+ atom(w, *it);
++it;
for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
+ atom(w, *it);
}
- b.append(']');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
}
-template <class T>
+template <class W, class T>
requires(std::is_same_v<T, std::string> ||
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
- b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ // Inline the escape path through the writer so we never round-trip
+ // pos through memory for string fields (Twitter is dominated by
+ // these -- sync/reload around each string was a real cost).
+ std::string_view input;
+ if constexpr (std::is_same_v<T, char>) {
+ input = std::string_view(&t, 1);
+ } else {
+ input = std::string_view(t);
+ }
+ // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+ // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * input.size())) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(input, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
}
-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
if (m.empty()) {
- b.append_raw("{}");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "{}", 2);
+ w.pos += 2;
return;
}
- b.append('{');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '{';
bool first = true;
for (const auto& [key, value] : m) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- // Keys must be convertible to string_view per the concept
- b.escape_and_append_with_quotes(key);
- b.append(':');
- atom(b, value);
+ // Keys must be convertible to string_view per the concept.
+ std::string_view key_sv(key);
+ // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+ // it wrapped to a small value, ensure() would spuriously succeed and the
+ // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+ // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+ // Note that this is pedantic except maybe on 32-bit targets.
+ if constexpr (W::checked) {
+ if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+ if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+ }
+ w.ptr[w.pos++] = '"';
+ w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+ w.ptr[w.pos++] = '"';
+ w.ptr[w.pos++] = ':';
+ atom(w, value);
}
- b.append('}');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '}';
}
-template<typename number_type,
+template<class W, typename number_type,
typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
- b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+ // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+ // path free of float-formatter machinery). For integers, write directly
+ // via jeaiii using local pos.
+ if constexpr (std::is_same_v<number_type, bool>) {
+ if (t) {
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "true", 4);
+ w.pos += 4;
+ } else {
+ if (!w.ensure(5)) return;
+ std::memcpy(w.ptr + w.pos, "false", 5);
+ w.pos += 5;
+ }
+ } else if constexpr (std::is_floating_point_v<number_type>) {
+ if constexpr (W::checked) {
+ call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+ } else {
+ w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+ }
+ } else if constexpr (std::is_unsigned_v<number_type>) {
+ if (!w.ensure(20)) return;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(t));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ } else {
+ // signed integral
+ if (!w.ensure(20)) return;
+ using U = typename std::make_unsigned<number_type>::type;
+ bool negative = t < 0;
+ U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+ w.ptr[w.pos] = '-';
+ w.pos += negative;
+ char *end = internal::write_uint_jeaiii(
+ w.ptr + w.pos, static_cast<uint64_t>(pv));
+ w.pos = static_cast<size_t>(end - w.ptr);
+ }
}
-template <class T>
+template <class W, class T>
requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
!concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> &&
@@ -59943,92 +73497,259 @@ template <class T>
!std::is_same_v<T, std::string_view> &&
!std::is_same_v<T, const char*> &&
!std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, t.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ // A transparent structure is serialized as its single member.
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ annotation_detail::atom_member<dm>(w, t.[:dm:]);
+ } else {
+ bool first = true;
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '{';
+ annotation_detail::atom_fields(w, t, first);
+ if (!w.ensure(1)) { return; }
+ w.ptr[w.pos++] = '}';
+ }
}
// Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
if (opt) {
- atom(b, opt.value());
+ atom(w, opt.value());
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
if (ptr) {
- atom(b, *ptr);
+ atom(w, *ptr);
} else {
- b.append_raw("null");
+ if (!w.ensure(4)) return;
+ std::memcpy(w.ptr + w.pos, "null", 4);
+ w.pos += 4;
}
}
// Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
#if SIMDJSON_STATIC_REFLECTION
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+ constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+ constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
if (e == [:enum_val:]) {
- b.append_raw(enum_str);
+ if (!w.ensure(enum_str_len)) return;
+ std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+ w.pos += enum_str_len;
return;
}
};
// Fallback to integer if enum value not found
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#else
// Fallback: serialize as integer if reflection not available
- atom(b, static_cast<std::underlying_type_t<T>>(e));
+ atom(w, static_cast<std::underlying_type_t<T>>(e));
#endif
}
// Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
if (container.empty()) {
- b.append_raw("[]");
+ if (!w.ensure(2)) return;
+ std::memcpy(w.ptr + w.pos, "[]", 2);
+ w.pos += 2;
return;
}
- b.append('[');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = '[';
bool first = true;
for (const auto& item : container) {
if (!first) {
- b.append(',');
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ',';
}
first = false;
- atom(b, item);
+ atom(w, item);
+ }
+ if (!w.ensure(1)) return;
+ w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+ if constexpr (require_custom_serialization<T>) {
+ return false;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+ return true;
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+ } else {
+ bool bounded = true;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+ is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+ }
+ };
+ return bounded;
+ }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+ size_t bound = 20; // the integer fallback
+ template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+ constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+ bound = (std::max)(bound, len);
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+ size_t bound = 0;
+ template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+ if constexpr (annotation_detail::is_serialized_member(dm)) {
+ if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+ bound += fields_bound(t.[:dm:]);
+ } else {
+ constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+ bound += rest_key_len + size_bound(t.[:dm:]);
+ }
+ }
+ };
+ return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+ if constexpr (std::is_same_v<T, char>) {
+ return 2 + 6;
+ } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+ std::is_same_v<T, const char *>) {
+ // Every byte may become \uXXXX, plus the quotes.
+ return 2 + 6 * std::string_view(t).size();
+ } else if constexpr (std::is_same_v<T, bool>) {
+ return 5;
+ } else if constexpr (std::is_floating_point_v<T>) {
+ return simdjson::internal::to_chars_buffer_size;
+ } else if constexpr (std::is_arithmetic_v<T>) {
+ return 20;
+ } else if constexpr (std::is_enum_v<T>) {
+ return enum_bound<T>();
+ } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+ return t ? size_bound(*t) : 4;
+ } else if constexpr (concepts::string_view_keyed_map<T>) {
+ size_t bound = 2;
+ for (const auto &[key, value] : t) {
+ // comma, quotes, colon
+ bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+ }
+ return bound;
+ } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+ using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+ if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+ // A fixed bound per element: no need to visit them.
+ return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+ } else {
+ size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+ for (const auto &item : t) {
+ bound += 1 + size_bound(item);
+ }
+ return bound;
+ }
+ } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+ constexpr auto dm = simdjson::detail::transparent_member(^^T);
+ return size_bound(t.[:dm:]);
+ } else {
+ return 2 + fields_bound(t);
+ }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+ // On 32-bit systems, the bound could overflow: keep the checked writer.
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+ const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+ const size_t pos = b.unsafe_position();
+ // The bound is a sum of in-memory sizes times a small constant: it cannot
+ // overflow on a 64-bit system. Be pedantic elsewhere.
+ if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+ const size_t cap = b.unsafe_capacity();
+ // Grow geometrically so that many small appends stay amortized.
+ if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+ unchecked_writer w(b.unsafe_data(), pos);
+ atom(w, t);
+ b.unsafe_set_position(w.pos);
+ }
+ return;
+ }
}
- b.append(']');
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
template <class T>
requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <class T>
@@ -60036,20 +73757,22 @@ template <class T>
std::is_same_v<T, std::string_view> ||
std::is_same_v<T, const char *> ||
std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ writer w(b);
+ atom(w, t);
+ w.sync();
}
template <concepts::optional_type T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::smart_pointer T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::appendable_containers T>
@@ -60057,14 +73780,14 @@ template <concepts::appendable_containers T>
!concepts::optional_type<T> && !concepts::smart_pointer<T> &&
!std::is_same_v<T, std::string> &&
!std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
template <concepts::string_view_keyed_map T>
requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
- atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+ append_bounded(b, t);
}
// works for struct
@@ -60078,39 +73801,15 @@ template <class Z>
!std::is_same_v<Z, std::string_view> &&
!std::is_same_v<Z, const char*> &&
!std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- int i = 0;
- b.append('{');
- template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
- if (i != 0)
- b.append(',');
- constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
- b.append_raw(key);
- b.append(':');
- atom(b, z.[:dm:]);
- i++;
- };
- b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
// works for container that have begin() and end() iterators
template <class Z>
requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
- auto it = z.begin();
- auto end = z.end();
- if (it == end) {
- b.append_raw("[]");
- return;
- }
- b.append('[');
- atom(b, *it);
- ++it;
- for (; it != end; ++it) {
- b.append(',');
- atom(b, *it);
- }
- b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+ append_bounded(b, z);
}
template <class Z>
@@ -60121,22 +73820,40 @@ void append(string_builder &b, const Z &z) {
template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+ // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+ // Prior related work: jsonifier's serializeJson resizes once through
+ // resize_and_overwrite (serializer.hpp).
+ (void)initial_capacity;
+ const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+ auto write = [&z](char *p) noexcept {
+ unchecked_writer w(p, 0);
+ atom(w, z);
+ return w.pos;
+ };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+ s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+ s.resize(bound);
+ s.resize(write(s.data()));
+#endif
+ return SUCCESS;
+ } else {
+ string_builder b(initial_capacity);
+ append(b, z);
+ std::string_view view;
+ if(auto e = b.view().get(view); e) { return e; }
+ s.assign(view);
+ return SUCCESS;
+ }
}
template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
- string_builder b(initial_capacity);
- append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+ std::string s;
+ if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+ return s;
}
template <class Z>
@@ -60149,40 +73866,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
template<constevalutil::fixed_string... FieldNames, typename T>
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
void extract_from(string_builder &b, const T &obj) {
- // Helper to check if a field name matches any of the requested fields
- auto should_extract = [](std::string_view field_name) constexpr -> bool {
- return ((FieldNames.view() == field_name) || ...);
- };
-
- b.append('{');
+ writer w(b);
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '{';
bool first = true;
-
// Iterate through all members of T using reflection
- template for (constexpr auto mem : std::define_static_array(
- std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+ static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+ template for (constexpr auto mem : members) {
if constexpr (std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+ static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
// Only serialize this field if it's in our list of requested fields
- if constexpr (should_extract(key)) {
- if (!first) {
- b.append(',');
+ if constexpr (((FieldNames.view() == key) || ...)) {
+ static constexpr auto first_key = std::define_static_string(
+ constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ static constexpr auto rest_key = std::define_static_string(
+ std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+ constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+ constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+ if (!w.ensure(rest_key_len)) { w.sync(); return; }
+ if (first) {
+ std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+ w.pos += first_key_len;
+ } else {
+ std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+ w.pos += rest_key_len;
}
first = false;
-
- // Serialize the key
- constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
- b.append_raw(quoted_key);
- b.append(':');
-
- // Serialize the value
- atom(b, obj.[:mem:]);
+ atom(w, obj.[:mem:]);
}
}
};
- b.append('}');
+ if (!w.ensure(1)) { w.sync(); return; }
+ w.ptr[w.pos++] = '}';
+ w.sync();
}
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -60195,25 +73913,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
return std::string(s);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+
} // namespace builder
} // namespace rvv_vls
// Alias the function template to 'to' in the global namespace
template <class Z>
simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = rvv_vls::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- rvv_vls::builder::string_builder b(initial_capacity);
- rvv_vls::builder::append(b, z);
- std::string_view s;
- if(auto e = b.view().get(s); e) { return e; }
- return std::string(s);
+ return rvv_vls::builder::to_json_string(z, initial_capacity);
}
template <class Z>
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = rvv_vls::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
- rvv_vls::builder::string_builder b(initial_capacity);
- rvv_vls::builder::append(b, z);
- std::string_view view;
- if(auto e = b.view().get(view); e) { return e; }
- s.assign(view);
- return SUCCESS;
+ return rvv_vls::builder::to_json(z, s, initial_capacity);
}
// Global namespace function for extract_from
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -60359,6 +74070,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
/* including simdjson/generic/builder/json_string_builder-inl.h for rvv_vls: #include "simdjson/generic/builder/json_string_builder-inl.h" */
/* begin file simdjson/generic/builder/json_string_builder-inl.h for rvv_vls */
#include <array>
+#include <cmath>
#include <cstring>
#include <limits>
#include <type_traits>
@@ -60393,6 +74105,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
#endif
#endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
#if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 && \
defined(__riscv_vector)
#ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -60412,6 +74129,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#endif
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
#include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
#ifdef _MSC_VER
#include <intrin.h>
#endif
@@ -60419,6 +74139,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
#if SIMDJSON_EXPERIMENTAL_HAS_LSX
#include <lsxintrin.h>
#endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
#if SIMDJSON_EXPERIMENTAL_HAS_RVV
#include <riscv_vector.h>
#endif
@@ -60468,105 +74191,6 @@ inline bool has_json_escapable_byte(uint64_t x) {
**/
-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
- for (char c : v) {
- // a table lookup is faster than a series of comparisons
- if (json_quotable_character[static_cast<uint8_t>(c)]) {
- return true;
- }
- }
- return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- uint8x16_t running = vdupq_n_u8(0);
- uint8x16_t v34 = vdupq_n_u8(34);
- uint8x16_t v92 = vdupq_n_u8(92);
-
- for (; i + 15 < view.size(); i += 16) {
- uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- if (i < view.size()) {
- uint8x16_t word =
- vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
- running = vorrq_u8(running, vceqq_u8(word, v34));
- running = vorrq_u8(running, vceqq_u8(word, v92));
- running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
- }
- return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __m128i running = _mm_setzero_si128();
- for (; i + 15 < view.size(); i += 16) {
-
- __m128i word =
- _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- if (i < view.size()) {
- __m128i word = _mm_loadu_si128(
- reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
- running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
- running = _mm_or_si128(
- running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
- _mm_setzero_si128()));
- }
- return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- if (view.size() < 16) {
- return simple_needs_escaping(view);
- }
- size_t i = 0;
- __vector unsigned char running = vec_splats((unsigned char)0);
- __vector unsigned char v34 = vec_splats((unsigned char)34);
- __vector unsigned char v92 = vec_splats((unsigned char)92);
- __vector unsigned char v32 = vec_splats((unsigned char)32);
-
- for (; i + 15 < view.size(); i += 16) {
- __vector unsigned char word =
- vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- if (i < view.size()) {
- __vector unsigned char word = vec_vsx_ld(
- 0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
- running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
- running = vec_or(running,
- (__vector unsigned char)vec_cmplt(word, v32));
- }
- return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
- return simple_needs_escaping(view);
-}
-#endif
-
// Scalar fallback for finding next quotable character
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
find_next_json_quotable_character_scalar(const std::string_view view,
@@ -60657,6 +74281,51 @@ find_next_json_quotable_character(const std::string_view view,
size_t current = len - remaining;
return find_next_json_quotable_character_scalar(view, current);
}
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+ size_t location) noexcept {
+ const size_t len = view.size();
+ const uint8_t *ptr =
+ reinterpret_cast<const uint8_t *>(view.data()) + location;
+ size_t remaining = len - location;
+
+ // SIMD constants for characters requiring escape
+ __m256i v34 = __lasx_xvreplgr2vr_b(34); // '"'
+ __m256i v92 = __lasx_xvreplgr2vr_b(92); // '\\'
+ __m256i v32 = __lasx_xvreplgr2vr_b(32); // control char threshold
+
+ while (remaining >= 32) {
+ __m256i word = __lasx_xvld(ptr, 0);
+
+ // Check for quotable characters: '"', '\\', or control chars (< 32)
+ __m256i needs_escape = __lasx_xvseq_b(word, v34);
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+ needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+ if (!__lasx_xbz_v(needs_escape)) {
+ // Found a quotable character - locate it via the four 64-bit lanes
+ uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+ uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+ uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+ uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+ size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+ if (lane0 != 0) {
+ return offset + trailing_zeroes(lane0) / 8;
+ } else if (lane1 != 0) {
+ return offset + 8 + trailing_zeroes(lane1) / 8;
+ } else if (lane2 != 0) {
+ return offset + 16 + trailing_zeroes(lane2) / 8;
+ } else {
+ return offset + 24 + trailing_zeroes(lane3) / 8;
+ }
+ }
+ ptr += 32;
+ remaining -= 32;
+ }
+ size_t current = len - remaining;
+ return find_next_json_quotable_character_scalar(view, current);
+}
#elif SIMDJSON_EXPERIMENTAL_HAS_LSX
simdjson_inline size_t
find_next_json_quotable_character(const std::string_view view,
@@ -60812,6 +74481,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
}
}
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ return _mm_unpacklo_epi64(
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+ _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ int32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ const __m128i v34 = _mm_set1_epi8(34); // '"'
+ const __m128i v92 = _mm_set1_epi8(92); // '\\'
+ const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+ __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+ needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+ return _mm_or_si128(
+ needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+ return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+ vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint64_t a64, b64;
+ memcpy(&a64, a, 8);
+ memcpy(&b64, b, 8);
+ return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+ const uint8_t *b) noexcept {
+ uint32_t a32, b32;
+ memcpy(&a32, a, 4);
+ memcpy(&b32, b, 4);
+ return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+ uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34)); // '"'
+ needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+ return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+ uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+ return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+ return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+ size_t n) noexcept {
+ if (n >= 8) {
+ memcpy(dst, src, 8);
+ memcpy(dst + n - 8, src + n - 8, 8);
+ } else if (n >= 4) {
+ memcpy(dst, src, 4);
+ memcpy(dst + n - 4, src + n - 4, 4);
+ } else if (n > 0) {
+ dst[0] = char(src[0]);
+ dst[n >> 1] = char(src[n >> 1]);
+ dst[n - 1] = char(src[n - 1]);
+ }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+ size_t i, size_t blockend,
+ uint64_t m) noexcept {
+ constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+ size_t pos = i; // first byte not yet copied
+ while (m) {
+ const size_t tz = trailing_zeroes(m);
+ const size_t next = i + tz / escape_mask_bits;
+ // Copy the run of safe bytes that precedes this escape.
+ copy_lt16(out, src + pos, next - pos);
+ out += next - pos;
+ escape_json_char(char(src[next]), out);
+ pos = next + 1;
+ m &= ~(lane << tz);
+ }
+ // Copy whatever follows the last escape.
+ copy_lt16(out, src + pos, blockend - pos);
+ return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+ const size_t len = input.size();
+ const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+ const char *const initout = out;
+
+ size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+ while (i + 32 <= len) {
+ const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+ const __m256i flags = _mm256_or_si256(
+ _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)), // '"'
+ _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))), // '\\'
+ _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+ _mm256_setzero_si256())); // control
+ const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+ if (simdjson_likely(mask == 0)) {
+ _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+ out += 32;
+ } else {
+ for (size_t half = 0; half < 32; half += 16) {
+ const uint64_t m = (mask >> half) & 0xFFFF;
+ if (m == 0) {
+ escape_store16(out, escape_load16(src + i + half));
+ out += 16;
+ } else {
+ out = escape_block(src, out, i + half, i + half + 16, m);
+ }
+ }
+ }
+ i += 32;
+ }
+#endif
+ while (i + 16 <= len) {
+ escape_vector word = escape_load16(src + i);
+ escape_vector flags = escape_flags(word);
+ if (simdjson_likely(!escape_any(flags))) {
+ escape_store16(out, word);
+ out += 16;
+ } else {
+ out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+ }
+ i += 16;
+ }
+ if (i < len) {
+ const size_t rem = len - i;
+ uint64_t m;
+ if (len >= 16) {
+ // The last 16 bytes of the input are in bounds. Bit k of that block's
+ // mask belongs to input position len - 16 + k, so shift it down to align
+ // bit 0 with position i.
+ m = escape_mask(escape_load16(src + len - 16)) >>
+ (escape_mask_bits * (16 - rem));
+ } else if (len >= 8) {
+ // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+ // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+ uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+ constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+ m = (mm & low8) |
+ ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+ } else if (len >= 4) {
+ // Same idea with two overlapping 4-byte loads.
+ uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+ constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+ m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+ << (escape_mask_bits * (len - 4)));
+ } else {
+ // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+ for (size_t k = 0; k < len; k++) {
+ uint8_t c = src[k];
+ if (json_quotable_character[c]) {
+ escape_json_char(char(c), out);
+ } else {
+ *out++ = char(c);
+ }
+ }
+ return size_t(out - initout);
+ }
+ if (m == 0) {
+ copy_lt16(out, src + i, rem);
+ out += rem;
+ } else {
+ out = escape_block(src, out, i, len, m);
+ }
+ }
+ return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
// Writes the escaped version of input to out, returning the number of bytes
// written. Uses SIMD position finding to locate quotable characters efficiently.
inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -60841,9 +74758,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
escape_json_char(input[location], out);
location += 1;
}
- return out - initout;
+ return size_t(out - initout);
}
+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
simdjson_inline string_builder::string_builder(size_t initial_capacity)
: buffer(new(std::nothrow) char[initial_capacity]), position(0),
capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -60866,7 +74787,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
return is_valid;
}
-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
if (!is_valid) {
return;
}
@@ -60922,81 +74843,136 @@ simdjson_inline void string_builder::clear() noexcept {
namespace internal {
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
- return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
- static uint64_t table[] = {
- 4294967296, 8589934582, 8589934582, 8589934582, 12884901788,
- 12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
- 21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
- 25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
- 34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
- 38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
- 42949672960, 42949672960};
- return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
- static uint64_t table[] = {9,
- 99,
- 999,
- 9999,
- 99999,
- 999999,
- 9999999,
- 99999999,
- 999999999,
- 9999999999,
- 99999999999,
- 999999999999,
- 9999999999999,
- 99999999999999,
- 999999999999999ULL,
- 9999999999999999ULL,
- 99999999999999999ULL,
- 999999999999999999ULL,
- 9999999999999999999ULL};
- int y = (19 * int_log2(x) >> 6);
- y += x > table[y];
- return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
- std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
- static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
- sizeof(number_type) == 2 || sizeof(number_type) == 1,
- "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
- SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
- return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+ "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+ "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+ "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+ "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+ "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+ "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+ "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+ std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+ constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ if (n < 100) {
+ jeaiii_write_fd(b, n);
+ return n < 10 ? b + 1 : b + 2;
+ }
+ if (n < 1000000) {
+ if (n < 10000) {
+ const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 24);
+ b -= n < 1000;
+ const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 24);
+ return b + 4;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 100000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ return b + 6;
+ }
+ const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+ jeaiii_write_fd(b, f0 >> 32);
+ b -= n < 10000000;
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+ constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+ const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+ jeaiii_write_dd(b, f0 >> 32);
+ const uint64_t f2 = (f0 & mask32) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 32);
+ const uint64_t f4 = (f2 & mask32) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 32);
+ const uint64_t f6 = (f4 & mask32) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 32);
+ return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+ constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+ const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+ jeaiii_write_fd(b, f0 >> 57);
+ b -= n < 1000000000;
+ const uint64_t f2 = (f0 & mask57) * 100;
+ jeaiii_write_dd(b + 2, f2 >> 57);
+ const uint64_t f4 = (f2 & mask57) * 100;
+ jeaiii_write_dd(b + 4, f4 >> 57);
+ const uint64_t f6 = (f4 & mask57) * 100;
+ jeaiii_write_dd(b + 6, f6 >> 57);
+ const uint64_t f8 = (f6 & mask57) * 100;
+ jeaiii_write_dd(b + 8, f8 >> 57);
+ return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+ if (n < 100000000) {
+ return jeaiii_lt1e8(b, uint32_t(n));
+ }
+ if (n < (uint64_t(1) << 32)) {
+ return jeaiii_9_or_10(b, n);
+ }
+ // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+ const uint32_t z = uint32_t(n % 100000000);
+ uint64_t u = n / 100000000;
+ if (u < 100000000) {
+ // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+ b = jeaiii_lt1e8(b, uint32_t(u));
+ } else if (u < (uint64_t(1) << 32)) {
+ b = jeaiii_9_or_10(b, u);
+ } else {
+ // u has 11 or 12 digits: split off 8 more.
+ const uint32_t y = uint32_t(u % 100000000);
+ u /= 100000000;
+ b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+ b = jeaiii_8_digits(b, y);
}
- else {
- return fast_digit_count_64(static_cast<uint64_t>(v));
- }
-}
-static const char decimal_table[200] = {
- 0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
- 0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
- 0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
- 0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
- 0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
- 0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
- 0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
- 0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
- 0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
- 0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
- 0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
- 0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
- 0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
- 0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
- 0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
- 0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
- 0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+ return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ std::memcpy(p, "NaN", 3);
+ return p + 3;
+ }
+ if (v < 0) {
+ *p++ = '-';
+ }
+ std::memcpy(p, "Infinity", 8);
+ return p + 8;
+ }
+#endif
+ return simdjson::internal::to_chars(p, nullptr, v);
+}
} // namespace internal
template <typename number_type, typename>
@@ -61024,87 +75000,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
- // Process 4 digits at a time instead of 2, reducing store operations
- // and divisions by approximately half for large numbers.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
- unsigned_type pv = static_cast<unsigned_type>(v);
- size_t dc = internal::digit_count(pv);
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100; // High 2 digits of remainder
- unsigned_type r_lo = r % 100; // Low 2 digits of remainder
- // Write low 2 digits first (rightmost), then high 2 digits
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits with original 2-digit loop
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position,
+ static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
- // Same 4-digit batching as unsigned path for signed integers
+ // 19 digits (max abs value of int64_t) + optional minus sign.
constexpr size_t max_number_size = 20;
if (capacity_check(max_number_size)) {
using unsigned_type = typename std::make_unsigned<number_type>::type;
bool negative = v < 0;
- unsigned_type pv = static_cast<unsigned_type>(v);
- if (negative) {
- pv = 0 - pv; // the 0 is for Microsoft
- }
- size_t dc = internal::digit_count(pv);
- // by always writing the minus sign, we avoid the branch.
+ // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+ unsigned_type pv = negative
+ ? unsigned_type(0) - static_cast<unsigned_type>(v)
+ : static_cast<unsigned_type>(v);
+ // Branchless: always write '-', advance only if negative.
buffer.get()[position] = '-';
- position += negative ? 1 : 0;
- char *write_pointer = buffer.get() + position + dc - 1;
-
- // Process 4 digits per iteration for large numbers
- while (pv >= 10000) {
- unsigned_type q = pv / 10000;
- unsigned_type r = pv % 10000;
- unsigned_type r_hi = r / 100;
- unsigned_type r_lo = r % 100;
- memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
- memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
- write_pointer -= 4;
- pv = q;
- }
-
- // Handle remaining 1-4 digits
- while (pv >= 100) {
- memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
- write_pointer -= 2;
- pv /= 100;
- }
- if (pv >= 10) {
- *write_pointer-- = char('0' + (pv % 10));
- pv /= 10;
- }
- *write_pointer = char('0' + pv);
- position += dc;
+ position += negative;
+ char* end = internal::write_uint_jeaiii(
+ buffer.get() + position, static_cast<uint64_t>(pv));
+ position = end - buffer.get();
}
}
else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
- constexpr size_t max_number_size = 24;
+ // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+ // but to_chars over-writes with fixed-size 16/17-byte copies so the
+ // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+ constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+ // Check if the input might be NaN or infinity
+ if (simdjson_unlikely(!std::isfinite(v))) {
+ if (std::isnan(v)) {
+ constexpr char nan_literal[] = "NaN";
+ constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+ std::memcpy(buffer.get() + position, nan_literal, nan_len);
+ position += nan_len;
+ } else {
+ constexpr char inf_literal[] = "Infinity";
+ constexpr size_t inf_len = sizeof(inf_literal) - 1;
+ if (v < 0) {
+ buffer.get()[position] = '-';
+ ++position;
+ }
+ std::memcpy(buffer.get() + position, inf_literal, inf_len);
+ position += inf_len;
+ }
+ return;
+ }
+#endif
+
// We could specialize for float.
char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
double(v));
@@ -61165,7 +75116,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
#endif
simdjson_inline void string_builder::append_raw(const char *c) noexcept {
- size_t len = std::strlen(c);
+ // char_traits::length is constexpr; lets the compiler fold the length
+ // when called with a pointer to a compile-time-constant string.
+ size_t len = std::char_traits<char>::length(c);
append_raw(c, len);
}
@@ -61184,6 +75137,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
position += len;
}
}
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+ if (capacity_check(N)) {
+ std::memcpy(buffer.get() + position, str, N);
+ position += N;
+ }
+}
#if SIMDJSON_SUPPORTS_CONCEPTS
// Support for optional types (std::optional, etc.)
template <concepts::optional_type T>
@@ -61213,7 +75174,7 @@ simdjson_inline void string_builder::append(const T &value) {
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
// Support for range-based appending (std::ranges::view, etc.)
template <std::ranges::range R>
- requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+ requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
simdjson_inline void string_builder::append(const R &range) noexcept {
auto it = std::ranges::begin(range);
auto end = std::ranges::end(range);
@@ -61448,10 +75409,14 @@ namespace simdjson {
// Otherwise, amalgamation will fail.
/* skipped duplicate #include "simdjson/dom/base.h" // for MINIMAL_DOCUMENT_CAPACITY */
/* skipped duplicate #include "simdjson/implementation.h" */
+/* skipped duplicate #include "simdjson/base.h" */
+/* skipped duplicate #include "simdjson/common_defs.h" */
+/* skipped duplicate #include "simdjson/constevalutil.h" */
/* skipped duplicate #include "simdjson/padded_string.h" */
/* skipped duplicate #include "simdjson/padded_string_view.h" */
/* skipped duplicate #include "simdjson/internal/dom_parser_implementation.h" */
/* skipped duplicate #include "simdjson/jsonpathutil.h" */
+/* skipped duplicate #include "simdjson/annotations.h" */
#endif // SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H
/* end file simdjson/generic/ondemand/dependencies.h */
@@ -61577,7 +75542,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {
/* result might be undefined when input_num is zero */
simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+ // if the system supports SVE or CSSC, __builtin_popcountll
+ // might be compiled to fewer single instructions. For CSSC,
+ // __builtin_popcountll is compiled to a single instruction.
+ return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
}
@@ -61614,15 +75586,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace arm64
@@ -61886,6 +75849,7 @@ namespace {
return vget_lane_u64(
vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
}
+ // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
};
@@ -62341,7 +76305,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/arm64/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for arm64: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for arm64 */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -62390,6 +76354,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace arm64
} // namespace simdjson
@@ -62422,6 +76393,9 @@ template <> struct is_builtin_deserializable<arm64::ondemand::object> : std::tru
template <> struct is_builtin_deserializable<arm64::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<arm64::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -62439,6 +76413,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = arm64::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = arm64::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = arm64::ondemand::array;
@@ -62653,6 +76631,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -62805,6 +76794,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -62823,6 +76814,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -62958,6 +76951,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -63026,9 +77028,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -63038,7 +77043,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -63053,7 +77058,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -63063,7 +77069,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -63091,7 +77097,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -63100,7 +77106,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -63178,6 +77185,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -63194,6 +77245,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -63221,6 +77319,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -63308,7 +77426,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -63773,9 +77891,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -63783,9 +77916,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -64116,6 +78259,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -64207,6 +78351,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -64231,6 +78378,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -65110,33 +79261,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -65264,6 +79469,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -65326,8 +79532,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -65458,7 +79675,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -65470,7 +79687,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -65526,6 +79743,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -65547,7 +79768,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<arm64::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -65567,7 +79789,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, arm64::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, arm64::ondemand::array>) {
return first;
@@ -65575,7 +79797,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, arm64::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, arm64::ondemand::array>) {
out = first;
@@ -65627,6 +79849,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -65669,6 +79900,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -65768,14 +80002,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -65813,6 +80047,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -65828,6 +80102,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -65841,6 +80162,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -65911,9 +80250,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -65922,7 +80264,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -65945,7 +80287,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -65957,7 +80299,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -65968,7 +80311,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -65981,7 +80324,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -65990,7 +80333,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -65999,7 +80343,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -66033,24 +80382,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -66060,7 +80409,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -66069,14 +80418,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -66571,9 +80920,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66585,7 +80949,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -66598,7 +80962,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -66610,7 +80975,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -66621,7 +80987,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -66634,7 +81000,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -66643,7 +81009,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -66652,7 +81019,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -66665,12 +81037,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -66732,9 +81104,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66743,11 +81130,31 @@ public:
simdjson_inline simdjson_result<arm64::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using arm64::implementation_simdjson_result_base<arm64::ondemand::document>::operator*;
@@ -66756,12 +81163,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator arm64::ondemand::array() & noexcept(false);
simdjson_inline operator arm64::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator arm64::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -66827,9 +81234,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66838,22 +81260,42 @@ public:
simdjson_inline simdjson_result<arm64::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator arm64::ondemand::array() & noexcept(false);
simdjson_inline operator arm64::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator arm64::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -67021,10 +81463,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -67033,6 +81472,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -67092,7 +81534,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -67146,13 +81591,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -67186,8 +81634,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -67201,6 +81664,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -67227,7 +81691,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -67302,6 +81766,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -67334,6 +81808,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -67365,11 +81849,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<arm64::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<arm64::ondemand::value> value() noexcept;
};
@@ -67377,6 +81867,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for arm64 */
+/* including simdjson/generic/ondemand/key_selector.h for arm64: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for arm64 */
/* including simdjson/generic/ondemand/object.h for arm64: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for arm64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -67386,6 +83268,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -67396,6 +83279,114 @@ namespace simdjson {
namespace arm64 {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -67414,8 +83405,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -67427,10 +83429,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -67503,6 +83506,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -67579,6 +83676,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -67624,7 +83749,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -67636,7 +83761,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -67688,10 +83813,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -67707,7 +83840,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<arm64::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<arm64::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<arm64::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -67725,6 +83859,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<arm64::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(arm64::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -67732,7 +83868,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, arm64::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, arm64::ondemand::object>) {
return first;
@@ -67740,7 +83876,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, arm64::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, arm64::ondemand::object>) {
out = first;
@@ -67750,6 +83886,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires arm64::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, arm64::ondemand::value>
+ simdjson_inline arm64::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, arm64::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires arm64::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline arm64::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline arm64::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -67790,6 +83959,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -67809,6 +83987,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -67854,6 +84035,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for arm64 */
+/* including simdjson/generic/ondemand/ranges.h for arm64: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::arm64::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::arm64::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for arm64 */
/* including simdjson/generic/ondemand/serialization.h for arm64: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for arm64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -67986,12 +84352,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -68016,10 +84384,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -68055,11 +84438,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -68083,22 +84514,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -68139,7 +84613,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -68158,21 +84632,21 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noe
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::value &val, T &out) noexcept(false) {
arm64::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::document &doc, T &out) noexcept(false) {
arm64::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc, T &out) noexcept(false) {
arm64::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -68183,10 +84657,6 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc,
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -68194,7 +84664,7 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc,
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -68206,12 +84676,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -68243,53 +84714,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, arm64::ondemand::number>
+&& !std::is_same_v<T, arm64::ondemand::document>
+&& !std::is_same_v<T, arm64::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^arm64::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = arm64::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, arm64::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ arm64::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ arm64::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ arm64::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ arm64::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, arm64::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
arm64::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, arm64::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, arm64::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, arm64::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -68305,33 +85318,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -68643,9 +85648,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -68672,6 +85685,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -68685,6 +85701,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -68692,31 +85711,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -68790,6 +85808,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -68819,10 +85840,14 @@ simdjson_inline simdjson_result<arm64::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -68885,6 +85910,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -68980,6 +86058,41 @@ namespace simdjson {
namespace arm64 {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -69011,6 +86124,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -69024,6 +86144,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -69037,17 +86163,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -69059,12 +86205,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -69072,12 +86232,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -69246,6 +86420,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -69283,6 +86460,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -69396,10 +86577,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::value>
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -69408,6 +86625,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondeman
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -69436,11 +86659,23 @@ template<> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::g
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -69710,16 +86945,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -69727,9 +86968,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -69751,11 +87019,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -69763,17 +87045,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -70112,6 +87412,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::docume
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -70120,10 +87436,36 @@ simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::documen
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -70151,22 +87493,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<arm64::ondemand::document>
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<arm64::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<arm64::ondemand::document>(first).get<T>(out);
}
@@ -70235,27 +87601,27 @@ simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::onde
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -70345,21 +87711,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -70371,11 +87754,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -70521,6 +87918,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::docume
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -70529,10 +87942,36 @@ simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::documen
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -70559,22 +87998,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<arm64::ondemand::document_
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<arm64::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<arm64::ondemand::document_reference>(first).get<T>(out);
}
@@ -70636,27 +88099,27 @@ simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator a
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator arm64::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator arm64::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -70722,6 +88185,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand:
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -70808,23 +88272,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -70833,6 +88294,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -70852,6 +88314,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -70932,13 +88397,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -71013,12 +88485,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -71042,10 +88571,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -71054,11 +88608,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -71066,14 +88628,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -71198,11 +88793,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -71224,6 +88827,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -71268,11 +88877,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondeman
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -71316,6 +88939,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -71327,6 +88953,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -71353,7 +88982,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -71425,7 +89055,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -71468,7 +89099,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -71485,6 +89116,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -71767,7 +89409,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -72106,6 +89748,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -72135,12 +89781,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -72150,6 +89805,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -72159,6 +89817,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -72194,6 +89996,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -72215,9 +90020,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -72226,7 +90039,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -72328,6 +90143,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -72335,9 +90153,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -72395,10 +90245,14 @@ simdjson_inline simdjson_result<arm64::ondemand::object>::simdjson_result(arm64:
simdjson_inline simdjson_result<arm64::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<arm64::ondemand::object>(error) {}
-simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -72452,11 +90306,55 @@ simdjson_inline error_code simdjson_result<arm64::ondemand::object>::for_each_at
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires arm64::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, arm64::ondemand::value>
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, arm64::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires arm64::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<arm64::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<arm64::ondemand::object_position> simdjson_result<arm64::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<arm64::ondemand::object>::revert_position(arm64::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<arm64::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -72500,6 +90398,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -72627,6 +90580,147 @@ simdjson_inline simdjson_result<arm64::ondemand::object_iterator> &simdjson_resu
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for arm64 */
+/* including simdjson/generic/ondemand/ranges-inl.h for arm64: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::arm64::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::arm64::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::arm64::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::arm64::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::arm64::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::arm64::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for arm64 */
/* including simdjson/generic/ondemand/parser-inl.h for arm64: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for arm64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -72658,7 +90752,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -72683,6 +90780,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -72699,6 +90797,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -72764,6 +90863,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -72771,8 +90898,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -72792,6 +90922,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -73199,6 +91374,27 @@ namespace simdjson {
namespace arm64 {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -73586,6 +91782,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -73703,7 +91915,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -73714,6 +91926,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -73747,6 +91968,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -73849,7 +92080,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -73861,6 +92092,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -73970,6 +92210,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -74198,6 +92475,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -75732,7 +94024,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
/* end file simdjson/fallback/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for fallback: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for fallback */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -75781,6 +94073,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace fallback
} // namespace simdjson
@@ -75813,6 +94112,9 @@ template <> struct is_builtin_deserializable<fallback::ondemand::object> : std::
template <> struct is_builtin_deserializable<fallback::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<fallback::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -75830,6 +94132,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = fallback::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = fallback::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = fallback::ondemand::array;
@@ -76044,6 +94350,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -76196,6 +94513,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -76214,6 +94533,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -76349,6 +94670,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -76417,9 +94747,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -76429,7 +94762,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -76444,7 +94777,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -76454,7 +94788,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -76482,7 +94816,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -76491,7 +94825,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -76569,6 +94904,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -76585,6 +94964,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -76612,6 +95038,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -76699,7 +95145,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -77164,9 +95610,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -77174,9 +95635,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -77507,6 +95978,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -77598,6 +96070,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -77622,6 +96097,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -78501,33 +96980,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -78655,6 +97188,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -78717,8 +97251,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -78849,7 +97394,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -78861,7 +97406,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -78917,6 +97462,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -78938,7 +97487,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<fallback::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -78958,7 +97508,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, fallback::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, fallback::ondemand::array>) {
return first;
@@ -78966,7 +97516,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, fallback::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, fallback::ondemand::array>) {
out = first;
@@ -79018,6 +97568,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -79060,6 +97619,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -79159,14 +97721,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -79204,6 +97766,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -79219,6 +97821,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -79232,6 +97881,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -79302,9 +97969,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -79313,7 +97983,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -79336,7 +98006,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -79348,7 +98018,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -79359,7 +98030,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -79372,7 +98043,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -79381,7 +98052,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -79390,7 +98062,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -79424,24 +98101,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -79451,7 +98128,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -79460,14 +98137,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -79962,9 +98639,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -79976,7 +98668,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -79989,7 +98681,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -80001,7 +98694,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -80012,7 +98706,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -80025,7 +98719,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -80034,7 +98728,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -80043,7 +98738,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -80056,12 +98756,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80123,9 +98823,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -80134,11 +98849,31 @@ public:
simdjson_inline simdjson_result<fallback::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using fallback::implementation_simdjson_result_base<fallback::ondemand::document>::operator*;
@@ -80147,12 +98882,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator fallback::ondemand::array() & noexcept(false);
simdjson_inline operator fallback::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator fallback::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80218,9 +98953,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -80229,22 +98979,42 @@ public:
simdjson_inline simdjson_result<fallback::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator fallback::ondemand::array() & noexcept(false);
simdjson_inline operator fallback::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator fallback::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80412,10 +99182,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -80424,6 +99191,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -80483,7 +99253,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -80537,13 +99310,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -80577,8 +99353,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -80592,6 +99383,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -80618,7 +99410,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -80693,6 +99485,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -80725,6 +99527,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -80756,11 +99568,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<fallback::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<fallback::ondemand::value> value() noexcept;
};
@@ -80768,6 +99586,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for fallback */
+/* including simdjson/generic/ondemand/key_selector.h for fallback: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for fallback */
/* including simdjson/generic/ondemand/object.h for fallback: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for fallback */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -80777,6 +100987,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -80787,6 +100998,114 @@ namespace simdjson {
namespace fallback {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -80805,8 +101124,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -80818,10 +101148,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -80894,6 +101225,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -80970,6 +101395,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -81015,7 +101468,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -81027,7 +101480,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -81079,10 +101532,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -81098,7 +101559,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<fallback::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<fallback::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<fallback::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -81116,6 +101578,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<fallback::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(fallback::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -81123,7 +101587,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, fallback::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, fallback::ondemand::object>) {
return first;
@@ -81131,7 +101595,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, fallback::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, fallback::ondemand::object>) {
out = first;
@@ -81141,6 +101605,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires fallback::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, fallback::ondemand::value>
+ simdjson_inline fallback::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, fallback::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires fallback::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline fallback::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline fallback::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -81181,6 +101678,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -81200,6 +101706,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -81245,6 +101754,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for fallback */
+/* including simdjson/generic/ondemand/ranges.h for fallback: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::fallback::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::fallback::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for fallback */
/* including simdjson/generic/ondemand/serialization.h for fallback: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for fallback */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -81377,12 +102071,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -81407,10 +102103,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -81446,11 +102157,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -81474,22 +102233,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -81530,7 +102332,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -81549,21 +102351,21 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out)
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::value &val, T &out) noexcept(false) {
fallback::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::document &doc, T &out) noexcept(false) {
fallback::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &doc, T &out) noexcept(false) {
fallback::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -81574,10 +102376,6 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &d
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -81585,7 +102383,7 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &d
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -81597,12 +102395,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -81634,53 +102433,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, fallback::ondemand::number>
+&& !std::is_same_v<T, fallback::ondemand::document>
+&& !std::is_same_v<T, fallback::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^fallback::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = fallback::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, fallback::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ fallback::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ fallback::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ fallback::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ fallback::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, fallback::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
fallback::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, fallback::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, fallback::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, fallback::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -81696,33 +103037,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -82034,9 +103367,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -82063,6 +103404,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -82076,6 +103420,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -82083,31 +103430,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -82181,6 +103527,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -82210,10 +103559,14 @@ simdjson_inline simdjson_result<fallback::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -82276,6 +103629,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -82371,6 +103777,41 @@ namespace simdjson {
namespace fallback {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -82402,6 +103843,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -82415,6 +103863,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -82428,17 +103882,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -82450,12 +103924,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -82463,12 +103951,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -82637,6 +104139,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -82674,6 +104179,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -82787,10 +104296,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::val
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -82799,6 +104344,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::onde
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -82827,11 +104378,23 @@ template<> simdjson_inline error_code simdjson_result<fallback::ondemand::value>
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -83101,16 +104664,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -83118,9 +104687,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -83142,11 +104738,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -83154,17 +104764,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -83503,6 +105131,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::doc
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -83511,10 +105155,36 @@ simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::docu
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -83542,22 +105212,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<fallback::ondemand::docume
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<fallback::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<fallback::ondemand::document>(first).get<T>(out);
}
@@ -83626,27 +105320,27 @@ simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -83736,21 +105430,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -83762,11 +105473,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -83912,6 +105637,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::doc
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -83920,10 +105661,36 @@ simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::docu
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -83950,22 +105717,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<fallback::ondemand::docume
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<fallback::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<fallback::ondemand::document_reference>(first).get<T>(out);
}
@@ -84027,27 +105818,27 @@ simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operato
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator fallback::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator fallback::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -84113,6 +105904,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondema
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -84199,23 +105991,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -84224,6 +106013,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -84243,6 +106033,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -84323,13 +106116,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -84404,12 +106204,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -84433,10 +106290,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -84445,11 +106327,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -84457,14 +106347,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -84589,11 +106512,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -84615,6 +106546,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -84659,11 +106596,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::onde
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -84707,6 +106658,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -84718,6 +106672,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -84744,7 +106701,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -84816,7 +106774,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -84859,7 +106818,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -84876,6 +106835,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -85158,7 +107128,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -85497,6 +107467,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -85526,12 +107500,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -85541,6 +107524,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -85550,6 +107536,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -85585,6 +107715,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -85606,9 +107739,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -85617,7 +107758,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -85719,6 +107862,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -85726,9 +107872,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -85786,10 +107964,14 @@ simdjson_inline simdjson_result<fallback::ondemand::object>::simdjson_result(fal
simdjson_inline simdjson_result<fallback::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<fallback::ondemand::object>(error) {}
-simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -85843,11 +108025,55 @@ simdjson_inline error_code simdjson_result<fallback::ondemand::object>::for_each
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires fallback::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, fallback::ondemand::value>
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, fallback::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires fallback::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<fallback::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<fallback::ondemand::object_position> simdjson_result<fallback::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<fallback::ondemand::object>::revert_position(fallback::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<fallback::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -85891,6 +108117,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -86018,6 +108299,147 @@ simdjson_inline simdjson_result<fallback::ondemand::object_iterator> &simdjson_r
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for fallback */
+/* including simdjson/generic/ondemand/ranges-inl.h for fallback: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::fallback::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::fallback::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::fallback::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::fallback::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::fallback::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::fallback::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for fallback */
/* including simdjson/generic/ondemand/parser-inl.h for fallback: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for fallback */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -86049,7 +108471,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -86074,6 +108499,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -86090,6 +108516,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -86155,6 +108582,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -86162,8 +108617,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -86183,6 +108641,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -86590,6 +109093,27 @@ namespace simdjson {
namespace fallback {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -86977,6 +109501,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -87094,7 +109634,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -87105,6 +109645,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -87138,6 +109687,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -87240,7 +109799,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -87252,6 +109811,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -87361,6 +109929,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -87589,6 +110194,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -89041,16 +111661,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace haswell
@@ -89610,7 +112220,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/haswell/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for haswell: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for haswell */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -89659,6 +112269,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace haswell
} // namespace simdjson
@@ -89691,6 +112308,9 @@ template <> struct is_builtin_deserializable<haswell::ondemand::object> : std::t
template <> struct is_builtin_deserializable<haswell::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<haswell::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -89708,6 +112328,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = haswell::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = haswell::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = haswell::ondemand::array;
@@ -89922,6 +112546,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -90074,6 +112709,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -90092,6 +112729,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -90227,6 +112866,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -90295,9 +112943,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -90307,7 +112958,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -90322,7 +112973,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -90332,7 +112984,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -90360,7 +113012,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -90369,7 +113021,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -90447,6 +113100,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -90463,6 +113160,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -90490,6 +113234,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -90577,7 +113341,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -91042,9 +113806,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -91052,9 +113831,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -91385,6 +114174,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -91476,6 +114266,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -91500,6 +114293,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -92379,33 +115176,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -92533,6 +115384,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -92595,8 +115447,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -92727,7 +115590,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -92739,7 +115602,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -92795,6 +115658,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -92816,7 +115683,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<haswell::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -92836,7 +115704,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, haswell::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, haswell::ondemand::array>) {
return first;
@@ -92844,7 +115712,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, haswell::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, haswell::ondemand::array>) {
out = first;
@@ -92896,6 +115764,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -92938,6 +115815,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -93037,14 +115917,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -93082,6 +115962,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -93097,6 +116017,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -93110,6 +116077,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -93180,9 +116165,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -93191,7 +116179,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -93214,7 +116202,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -93226,7 +116214,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -93237,7 +116226,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -93250,7 +116239,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -93259,7 +116248,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -93268,7 +116258,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -93302,24 +116297,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -93329,7 +116324,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -93338,14 +116333,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -93840,9 +116835,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -93854,7 +116864,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -93867,7 +116877,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -93879,7 +116890,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -93890,7 +116902,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -93903,7 +116915,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -93912,7 +116924,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -93921,7 +116934,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -93934,12 +116952,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94001,9 +117019,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -94012,11 +117045,31 @@ public:
simdjson_inline simdjson_result<haswell::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using haswell::implementation_simdjson_result_base<haswell::ondemand::document>::operator*;
@@ -94025,12 +117078,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator haswell::ondemand::array() & noexcept(false);
simdjson_inline operator haswell::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator haswell::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94096,9 +117149,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -94107,22 +117175,42 @@ public:
simdjson_inline simdjson_result<haswell::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator haswell::ondemand::array() & noexcept(false);
simdjson_inline operator haswell::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator haswell::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94290,10 +117378,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -94302,6 +117387,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -94361,7 +117449,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -94415,13 +117506,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -94455,8 +117549,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -94470,6 +117579,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -94496,7 +117606,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -94571,6 +117681,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -94603,6 +117723,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -94634,11 +117764,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<haswell::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<haswell::ondemand::value> value() noexcept;
};
@@ -94646,6 +117782,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for haswell */
+/* including simdjson/generic/ondemand/key_selector.h for haswell: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for haswell */
/* including simdjson/generic/ondemand/object.h for haswell: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for haswell */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -94655,6 +119183,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -94665,6 +119194,114 @@ namespace simdjson {
namespace haswell {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -94683,8 +119320,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -94696,10 +119344,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -94772,6 +119421,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -94848,6 +119591,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -94893,7 +119664,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -94905,7 +119676,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -94957,10 +119728,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -94976,7 +119755,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<haswell::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<haswell::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<haswell::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -94994,6 +119774,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<haswell::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(haswell::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -95001,7 +119783,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, haswell::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, haswell::ondemand::object>) {
return first;
@@ -95009,7 +119791,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, haswell::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, haswell::ondemand::object>) {
out = first;
@@ -95019,6 +119801,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires haswell::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, haswell::ondemand::value>
+ simdjson_inline haswell::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, haswell::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires haswell::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline haswell::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline haswell::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -95059,6 +119874,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -95078,6 +119902,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -95123,6 +119950,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for haswell */
+/* including simdjson/generic/ondemand/ranges.h for haswell: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::haswell::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::haswell::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for haswell */
/* including simdjson/generic/ondemand/serialization.h for haswell: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for haswell */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -95255,12 +120267,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -95285,10 +120299,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -95324,11 +120353,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -95352,22 +120429,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -95408,7 +120528,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -95427,21 +120547,21 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) n
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::value &val, T &out) noexcept(false) {
haswell::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::document &doc, T &out) noexcept(false) {
haswell::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &doc, T &out) noexcept(false) {
haswell::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -95452,10 +120572,6 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &do
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -95463,7 +120579,7 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &do
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -95475,12 +120591,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -95512,53 +120629,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, haswell::ondemand::number>
+&& !std::is_same_v<T, haswell::ondemand::document>
+&& !std::is_same_v<T, haswell::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^haswell::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = haswell::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, haswell::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ haswell::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ haswell::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ haswell::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ haswell::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, haswell::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
haswell::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, haswell::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, haswell::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, haswell::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -95574,33 +121233,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -95912,9 +121563,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -95941,6 +121600,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -95954,6 +121616,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -95961,31 +121626,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -96059,6 +121723,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -96088,10 +121755,14 @@ simdjson_inline simdjson_result<haswell::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -96154,6 +121825,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -96249,6 +121973,41 @@ namespace simdjson {
namespace haswell {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -96280,6 +122039,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -96293,6 +122059,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -96306,17 +122078,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -96328,12 +122120,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -96341,12 +122147,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -96515,6 +122335,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -96552,6 +122375,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -96665,10 +122492,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::valu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -96677,6 +122540,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondem
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -96705,11 +122574,23 @@ template<> simdjson_inline error_code simdjson_result<haswell::ondemand::value>:
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -96979,16 +122860,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -96996,9 +122883,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -97020,11 +122934,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -97032,17 +122960,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -97381,6 +123327,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -97389,10 +123351,36 @@ simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -97420,22 +123408,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<haswell::ondemand::documen
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<haswell::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<haswell::ondemand::document>(first).get<T>(out);
}
@@ -97504,27 +123516,27 @@ simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -97614,21 +123626,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -97640,11 +123669,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -97790,6 +123833,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -97798,10 +123857,36 @@ simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -97828,22 +123913,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<haswell::ondemand::documen
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<haswell::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<haswell::ondemand::document_reference>(first).get<T>(out);
}
@@ -97905,27 +124014,27 @@ simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator haswell::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator haswell::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -97991,6 +124100,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondeman
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -98077,23 +124187,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -98102,6 +124209,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -98121,6 +124229,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -98201,13 +124312,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -98282,12 +124400,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -98311,10 +124486,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -98323,11 +124523,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -98335,14 +124543,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -98467,11 +124708,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -98493,6 +124742,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -98537,11 +124792,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondem
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -98585,6 +124854,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -98596,6 +124868,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -98622,7 +124897,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -98694,7 +124970,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -98737,7 +125014,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -98754,6 +125031,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -99036,7 +125324,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -99375,6 +125663,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -99404,12 +125696,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -99419,6 +125720,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -99428,6 +125732,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -99463,6 +125911,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -99484,9 +125935,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -99495,7 +125954,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -99597,6 +126058,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -99604,9 +126068,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -99664,10 +126160,14 @@ simdjson_inline simdjson_result<haswell::ondemand::object>::simdjson_result(hasw
simdjson_inline simdjson_result<haswell::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<haswell::ondemand::object>(error) {}
-simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -99721,11 +126221,55 @@ simdjson_inline error_code simdjson_result<haswell::ondemand::object>::for_each_
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires haswell::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, haswell::ondemand::value>
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, haswell::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires haswell::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<haswell::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<haswell::ondemand::object_position> simdjson_result<haswell::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<haswell::ondemand::object>::revert_position(haswell::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<haswell::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -99769,6 +126313,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -99896,6 +126495,147 @@ simdjson_inline simdjson_result<haswell::ondemand::object_iterator> &simdjson_re
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for haswell */
+/* including simdjson/generic/ondemand/ranges-inl.h for haswell: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::haswell::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::haswell::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::haswell::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::haswell::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::haswell::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::haswell::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for haswell */
/* including simdjson/generic/ondemand/parser-inl.h for haswell: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for haswell */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -99927,7 +126667,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -99952,6 +126695,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -99968,6 +126712,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -100033,6 +126778,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -100040,8 +126813,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -100061,6 +126837,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -100468,6 +127289,27 @@ namespace simdjson {
namespace haswell {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -100855,6 +127697,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -100972,7 +127830,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -100983,6 +127841,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -101016,6 +127883,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -101118,7 +127995,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -101130,6 +128007,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -101239,6 +128125,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -101467,6 +128390,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -102916,16 +129854,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace icelake
@@ -103488,7 +130416,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
/* end file simdjson/icelake/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for icelake: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for icelake */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -103537,6 +130465,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace icelake
} // namespace simdjson
@@ -103569,6 +130504,9 @@ template <> struct is_builtin_deserializable<icelake::ondemand::object> : std::t
template <> struct is_builtin_deserializable<icelake::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<icelake::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -103586,6 +130524,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = icelake::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = icelake::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = icelake::ondemand::array;
@@ -103800,6 +130742,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -103952,6 +130905,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -103970,6 +130925,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -104105,6 +131062,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -104173,9 +131139,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -104185,7 +131154,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -104200,7 +131169,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -104210,7 +131180,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -104238,7 +131208,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -104247,7 +131217,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -104325,6 +131296,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -104341,6 +131356,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -104368,6 +131430,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -104455,7 +131537,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -104920,9 +132002,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -104930,9 +132027,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -105263,6 +132370,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -105354,6 +132462,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -105378,6 +132489,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -106257,33 +133372,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -106411,6 +133580,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -106473,8 +133643,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -106605,7 +133786,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -106617,7 +133798,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -106673,6 +133854,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -106694,7 +133879,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<icelake::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -106714,7 +133900,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, icelake::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, icelake::ondemand::array>) {
return first;
@@ -106722,7 +133908,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, icelake::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, icelake::ondemand::array>) {
out = first;
@@ -106774,6 +133960,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -106816,6 +134011,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -106915,14 +134113,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -106960,6 +134158,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -106975,6 +134213,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -106988,6 +134273,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -107058,9 +134361,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -107069,7 +134375,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -107092,7 +134398,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -107104,7 +134410,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -107115,7 +134422,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -107128,7 +134435,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -107137,7 +134444,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -107146,7 +134454,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -107180,24 +134493,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -107207,7 +134520,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -107216,14 +134529,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -107718,9 +135031,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107732,7 +135060,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -107745,7 +135073,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -107757,7 +135086,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -107768,7 +135098,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -107781,7 +135111,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -107790,7 +135120,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -107799,7 +135130,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -107812,12 +135148,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -107879,9 +135215,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107890,11 +135241,31 @@ public:
simdjson_inline simdjson_result<icelake::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using icelake::implementation_simdjson_result_base<icelake::ondemand::document>::operator*;
@@ -107903,12 +135274,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator icelake::ondemand::array() & noexcept(false);
simdjson_inline operator icelake::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator icelake::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -107974,9 +135345,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107985,22 +135371,42 @@ public:
simdjson_inline simdjson_result<icelake::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator icelake::ondemand::array() & noexcept(false);
simdjson_inline operator icelake::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator icelake::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -108168,10 +135574,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -108180,6 +135583,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -108239,7 +135645,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -108293,13 +135702,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -108333,8 +135745,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -108348,6 +135775,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -108374,7 +135802,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -108449,6 +135877,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -108481,6 +135919,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -108512,11 +135960,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<icelake::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<icelake::ondemand::value> value() noexcept;
};
@@ -108524,6 +135978,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for icelake */
+/* including simdjson/generic/ondemand/key_selector.h for icelake: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for icelake */
/* including simdjson/generic/ondemand/object.h for icelake: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for icelake */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -108533,6 +137379,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -108543,6 +137390,114 @@ namespace simdjson {
namespace icelake {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -108561,8 +137516,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -108574,10 +137540,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -108650,6 +137617,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -108726,6 +137787,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -108771,7 +137860,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -108783,7 +137872,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -108835,10 +137924,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -108854,7 +137951,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<icelake::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<icelake::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<icelake::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -108872,6 +137970,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<icelake::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(icelake::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -108879,7 +137979,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, icelake::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, icelake::ondemand::object>) {
return first;
@@ -108887,7 +137987,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, icelake::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, icelake::ondemand::object>) {
out = first;
@@ -108897,6 +137997,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires icelake::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, icelake::ondemand::value>
+ simdjson_inline icelake::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, icelake::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires icelake::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline icelake::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline icelake::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -108937,6 +138070,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -108956,6 +138098,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -109001,6 +138146,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for icelake */
+/* including simdjson/generic/ondemand/ranges.h for icelake: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::icelake::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::icelake::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for icelake */
/* including simdjson/generic/ondemand/serialization.h for icelake: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for icelake */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -109133,12 +138463,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -109163,10 +138495,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -109202,11 +138549,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -109230,22 +138625,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -109286,7 +138724,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -109305,21 +138743,21 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) n
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::value &val, T &out) noexcept(false) {
icelake::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::document &doc, T &out) noexcept(false) {
icelake::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &doc, T &out) noexcept(false) {
icelake::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -109330,10 +138768,6 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &do
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -109341,7 +138775,7 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &do
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -109353,12 +138787,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -109390,53 +138825,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, icelake::ondemand::number>
+&& !std::is_same_v<T, icelake::ondemand::document>
+&& !std::is_same_v<T, icelake::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^icelake::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = icelake::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, icelake::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ icelake::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ icelake::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ icelake::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ icelake::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, icelake::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
icelake::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, icelake::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, icelake::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, icelake::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -109452,33 +139429,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -109790,9 +139759,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -109819,6 +139796,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -109832,6 +139812,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -109839,31 +139822,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -109937,6 +139919,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -109966,10 +139951,14 @@ simdjson_inline simdjson_result<icelake::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -110032,6 +140021,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -110127,6 +140169,41 @@ namespace simdjson {
namespace icelake {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -110158,6 +140235,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -110171,6 +140255,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -110184,17 +140274,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -110206,12 +140316,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -110219,12 +140343,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -110393,6 +140531,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -110430,6 +140571,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -110543,10 +140688,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::valu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -110555,6 +140736,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondem
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -110583,11 +140770,23 @@ template<> simdjson_inline error_code simdjson_result<icelake::ondemand::value>:
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -110857,16 +141056,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -110874,9 +141079,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -110898,11 +141130,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -110910,17 +141156,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -111259,6 +141523,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -111267,10 +141547,36 @@ simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -111298,22 +141604,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<icelake::ondemand::documen
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<icelake::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<icelake::ondemand::document>(first).get<T>(out);
}
@@ -111382,27 +141712,27 @@ simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -111492,21 +141822,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -111518,11 +141865,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -111668,6 +142029,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -111676,10 +142053,36 @@ simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -111706,22 +142109,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<icelake::ondemand::documen
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<icelake::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<icelake::ondemand::document_reference>(first).get<T>(out);
}
@@ -111783,27 +142210,27 @@ simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator icelake::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator icelake::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -111869,6 +142296,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondeman
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -111955,23 +142383,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -111980,6 +142405,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -111999,6 +142425,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -112079,13 +142508,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -112160,12 +142596,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -112189,10 +142682,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -112201,11 +142719,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -112213,14 +142739,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -112345,11 +142904,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -112371,6 +142938,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -112415,11 +142988,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondem
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -112463,6 +143050,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -112474,6 +143064,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -112500,7 +143093,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -112572,7 +143166,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -112615,7 +143210,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -112632,6 +143227,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -112914,7 +143520,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -113253,6 +143859,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -113282,12 +143892,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -113297,6 +143916,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -113306,6 +143928,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -113341,6 +144107,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -113362,9 +144131,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -113373,7 +144150,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -113475,6 +144254,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -113482,9 +144264,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -113542,10 +144356,14 @@ simdjson_inline simdjson_result<icelake::ondemand::object>::simdjson_result(icel
simdjson_inline simdjson_result<icelake::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<icelake::ondemand::object>(error) {}
-simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -113599,11 +144417,55 @@ simdjson_inline error_code simdjson_result<icelake::ondemand::object>::for_each_
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires icelake::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, icelake::ondemand::value>
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, icelake::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires icelake::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<icelake::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<icelake::ondemand::object_position> simdjson_result<icelake::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<icelake::ondemand::object>::revert_position(icelake::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<icelake::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -113647,6 +144509,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -113774,6 +144691,147 @@ simdjson_inline simdjson_result<icelake::ondemand::object_iterator> &simdjson_re
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for icelake */
+/* including simdjson/generic/ondemand/ranges-inl.h for icelake: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::icelake::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::icelake::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::icelake::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::icelake::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::icelake::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::icelake::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for icelake */
/* including simdjson/generic/ondemand/parser-inl.h for icelake: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for icelake */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -113805,7 +144863,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -113830,6 +144891,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -113846,6 +144908,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -113911,6 +144974,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -113918,8 +145009,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -113939,6 +145033,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -114346,6 +145485,27 @@ namespace simdjson {
namespace icelake {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -114733,6 +145893,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -114850,7 +146026,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -114861,6 +146037,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -114894,6 +146079,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -114996,7 +146191,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -115008,6 +146203,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -115117,6 +146321,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -115345,6 +146586,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -116766,16 +148022,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- *result = value1 + value2;
- return *result < value1;
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace ppc64
@@ -117481,7 +148727,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/ppc64/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for ppc64: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for ppc64 */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -117530,6 +148776,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace ppc64
} // namespace simdjson
@@ -117562,6 +148815,9 @@ template <> struct is_builtin_deserializable<ppc64::ondemand::object> : std::tru
template <> struct is_builtin_deserializable<ppc64::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<ppc64::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -117579,6 +148835,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = ppc64::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = ppc64::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = ppc64::ondemand::array;
@@ -117793,6 +149053,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -117945,6 +149216,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -117963,6 +149236,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -118098,6 +149373,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -118166,9 +149450,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -118178,7 +149465,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -118193,7 +149480,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -118203,7 +149491,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -118231,7 +149519,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -118240,7 +149528,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -118318,6 +149607,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -118334,6 +149667,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -118361,6 +149741,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -118448,7 +149848,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -118913,9 +150313,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -118923,9 +150338,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -119256,6 +150681,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -119347,6 +150773,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -119371,6 +150800,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -120250,33 +151683,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -120404,6 +151891,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -120466,8 +151954,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -120598,7 +152097,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -120610,7 +152109,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -120666,6 +152165,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -120687,7 +152190,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -120707,7 +152211,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, ppc64::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, ppc64::ondemand::array>) {
return first;
@@ -120715,7 +152219,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, ppc64::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, ppc64::ondemand::array>) {
out = first;
@@ -120767,6 +152271,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -120809,6 +152322,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -120908,14 +152424,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -120953,6 +152469,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -120968,6 +152524,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -120981,6 +152584,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -121051,9 +152672,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -121062,7 +152686,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -121085,7 +152709,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -121097,7 +152721,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -121108,7 +152733,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -121121,7 +152746,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -121130,7 +152755,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -121139,7 +152765,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -121173,24 +152804,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -121200,7 +152831,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -121209,14 +152840,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -121711,9 +153342,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121725,7 +153371,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -121738,7 +153384,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -121750,7 +153397,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -121761,7 +153409,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -121774,7 +153422,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -121783,7 +153431,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -121792,7 +153441,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -121805,12 +153459,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -121872,9 +153526,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121883,11 +153552,31 @@ public:
simdjson_inline simdjson_result<ppc64::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using ppc64::implementation_simdjson_result_base<ppc64::ondemand::document>::operator*;
@@ -121896,12 +153585,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator ppc64::ondemand::array() & noexcept(false);
simdjson_inline operator ppc64::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator ppc64::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -121967,9 +153656,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121978,22 +153682,42 @@ public:
simdjson_inline simdjson_result<ppc64::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator ppc64::ondemand::array() & noexcept(false);
simdjson_inline operator ppc64::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator ppc64::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -122161,10 +153885,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -122173,6 +153894,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -122232,7 +153956,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -122286,13 +154013,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -122326,8 +154056,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -122341,6 +154086,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -122367,7 +154113,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -122442,6 +154188,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -122474,6 +154230,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -122505,11 +154271,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<ppc64::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<ppc64::ondemand::value> value() noexcept;
};
@@ -122517,6 +154289,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for ppc64 */
+/* including simdjson/generic/ondemand/key_selector.h for ppc64: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for ppc64 */
/* including simdjson/generic/ondemand/object.h for ppc64: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for ppc64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -122526,6 +155690,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -122536,6 +155701,114 @@ namespace simdjson {
namespace ppc64 {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -122554,8 +155827,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -122567,10 +155851,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -122643,6 +155928,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -122719,6 +156098,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -122764,7 +156171,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -122776,7 +156183,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -122828,10 +156235,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -122847,7 +156262,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<ppc64::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<ppc64::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -122865,6 +156281,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<ppc64::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(ppc64::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -122872,7 +156290,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, ppc64::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, ppc64::ondemand::object>) {
return first;
@@ -122880,7 +156298,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, ppc64::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, ppc64::ondemand::object>) {
out = first;
@@ -122890,6 +156308,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires ppc64::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, ppc64::ondemand::value>
+ simdjson_inline ppc64::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, ppc64::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires ppc64::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline ppc64::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline ppc64::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -122930,6 +156381,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -122949,6 +156409,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -122994,6 +156457,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for ppc64 */
+/* including simdjson/generic/ondemand/ranges.h for ppc64: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::ppc64::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::ppc64::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for ppc64 */
/* including simdjson/generic/ondemand/serialization.h for ppc64: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for ppc64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -123126,12 +156774,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -123156,10 +156806,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -123195,11 +156860,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -123223,22 +156936,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -123279,7 +157035,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -123298,21 +157054,21 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noe
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::value &val, T &out) noexcept(false) {
ppc64::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::document &doc, T &out) noexcept(false) {
ppc64::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc, T &out) noexcept(false) {
ppc64::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -123323,10 +157079,6 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc,
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -123334,7 +157086,7 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc,
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -123346,12 +157098,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -123383,53 +157136,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, ppc64::ondemand::number>
+&& !std::is_same_v<T, ppc64::ondemand::document>
+&& !std::is_same_v<T, ppc64::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^ppc64::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = ppc64::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, ppc64::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ ppc64::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ ppc64::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ ppc64::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ ppc64::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, ppc64::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
ppc64::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, ppc64::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, ppc64::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, ppc64::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -123445,33 +157740,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -123783,9 +158070,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -123812,6 +158107,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -123825,6 +158123,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -123832,31 +158133,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -123930,6 +158230,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -123959,10 +158262,14 @@ simdjson_inline simdjson_result<ppc64::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -124025,6 +158332,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -124120,6 +158480,41 @@ namespace simdjson {
namespace ppc64 {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -124151,6 +158546,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -124164,6 +158566,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -124177,17 +158585,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -124199,12 +158627,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -124212,12 +158654,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -124386,6 +158842,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -124423,6 +158882,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -124536,10 +158999,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::value>
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -124548,6 +159047,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondeman
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -124576,11 +159081,23 @@ template<> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::g
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -124850,16 +159367,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -124867,9 +159390,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -124891,11 +159441,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -124903,17 +159467,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -125252,6 +159834,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::docume
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -125260,10 +159858,36 @@ simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::documen
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -125291,22 +159915,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<ppc64::ondemand::document>
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<ppc64::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<ppc64::ondemand::document>(first).get<T>(out);
}
@@ -125375,27 +160023,27 @@ simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::onde
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -125485,21 +160133,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -125511,11 +160176,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -125661,6 +160340,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::docume
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -125669,10 +160364,36 @@ simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::documen
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -125699,22 +160420,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<ppc64::ondemand::document_
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<ppc64::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<ppc64::ondemand::document_reference>(first).get<T>(out);
}
@@ -125776,27 +160521,27 @@ simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator p
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator ppc64::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator ppc64::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -125862,6 +160607,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand:
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -125948,23 +160694,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -125973,6 +160716,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -125992,6 +160736,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -126072,13 +160819,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -126153,12 +160907,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -126182,10 +160993,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -126194,11 +161030,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -126206,14 +161050,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -126338,11 +161215,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -126364,6 +161249,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -126408,11 +161299,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondeman
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -126456,6 +161361,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -126467,6 +161375,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -126493,7 +161404,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -126565,7 +161477,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -126608,7 +161521,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -126625,6 +161538,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -126907,7 +161831,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -127246,6 +162170,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -127275,12 +162203,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -127290,6 +162227,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -127299,6 +162239,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -127334,6 +162418,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -127355,9 +162442,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -127366,7 +162461,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -127468,6 +162565,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -127475,9 +162575,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -127535,10 +162667,14 @@ simdjson_inline simdjson_result<ppc64::ondemand::object>::simdjson_result(ppc64:
simdjson_inline simdjson_result<ppc64::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<ppc64::ondemand::object>(error) {}
-simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -127592,11 +162728,55 @@ simdjson_inline error_code simdjson_result<ppc64::ondemand::object>::for_each_at
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires ppc64::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, ppc64::ondemand::value>
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, ppc64::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires ppc64::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<ppc64::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<ppc64::ondemand::object_position> simdjson_result<ppc64::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<ppc64::ondemand::object>::revert_position(ppc64::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<ppc64::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -127640,6 +162820,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -127767,6 +163002,147 @@ simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> &simdjson_resu
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for ppc64 */
+/* including simdjson/generic/ondemand/ranges-inl.h for ppc64: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::ppc64::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::ppc64::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::ppc64::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::ppc64::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::ppc64::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::ppc64::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for ppc64 */
/* including simdjson/generic/ondemand/parser-inl.h for ppc64: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for ppc64 */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -127798,7 +163174,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -127823,6 +163202,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -127839,6 +163219,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -127904,6 +163285,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -127911,8 +163320,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -127932,6 +163344,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -128339,6 +163796,27 @@ namespace simdjson {
namespace ppc64 {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -128726,6 +164204,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -128843,7 +164337,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -128854,6 +164348,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -128887,6 +164390,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -128989,7 +164502,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -129001,6 +164514,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -129110,6 +164632,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -129338,6 +164897,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -130773,16 +166347,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -131361,16 +166925,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
}
#endif
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
- return _addcarry_u64(0, value1, value2,
- reinterpret_cast<unsigned __int64 *>(result));
-#else
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-#endif
-}
} // unnamed namespace
} // namespace westmere
@@ -131791,7 +167345,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/westmere/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for westmere: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for westmere */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -131840,6 +167394,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace westmere
} // namespace simdjson
@@ -131872,6 +167433,9 @@ template <> struct is_builtin_deserializable<westmere::ondemand::object> : std::
template <> struct is_builtin_deserializable<westmere::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<westmere::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -131889,6 +167453,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = westmere::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = westmere::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = westmere::ondemand::array;
@@ -132103,6 +167671,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -132255,6 +167834,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -132273,6 +167854,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -132408,6 +167991,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -132476,9 +168068,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -132488,7 +168083,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -132503,7 +168098,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -132513,7 +168109,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -132541,7 +168137,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -132550,7 +168146,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -132628,6 +168225,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -132644,6 +168285,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -132671,6 +168359,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -132758,7 +168466,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -133223,9 +168931,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -133233,9 +168956,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -133566,6 +169299,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -133657,6 +169391,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -133681,6 +169418,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -134560,33 +170301,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -134714,6 +170509,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -134776,8 +170572,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -134908,7 +170715,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -134920,7 +170727,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -134976,6 +170783,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -134997,7 +170808,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<westmere::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -135017,7 +170829,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, westmere::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, westmere::ondemand::array>) {
return first;
@@ -135025,7 +170837,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, westmere::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, westmere::ondemand::array>) {
out = first;
@@ -135077,6 +170889,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -135119,6 +170940,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -135218,14 +171042,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -135263,6 +171087,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -135278,6 +171142,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -135291,6 +171202,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -135361,9 +171290,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -135372,7 +171304,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -135395,7 +171327,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -135407,7 +171339,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -135418,7 +171351,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -135431,7 +171364,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -135440,7 +171373,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -135449,7 +171383,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -135483,24 +171422,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -135510,7 +171449,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -135519,14 +171458,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -136021,9 +171960,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136035,7 +171989,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -136048,7 +172002,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -136060,7 +172015,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -136071,7 +172027,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -136084,7 +172040,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -136093,7 +172049,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -136102,7 +172059,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -136115,12 +172077,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136182,9 +172144,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136193,11 +172170,31 @@ public:
simdjson_inline simdjson_result<westmere::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using westmere::implementation_simdjson_result_base<westmere::ondemand::document>::operator*;
@@ -136206,12 +172203,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator westmere::ondemand::array() & noexcept(false);
simdjson_inline operator westmere::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator westmere::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136277,9 +172274,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136288,22 +172300,42 @@ public:
simdjson_inline simdjson_result<westmere::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator westmere::ondemand::array() & noexcept(false);
simdjson_inline operator westmere::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator westmere::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136471,10 +172503,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -136483,6 +172512,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -136542,7 +172574,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -136596,13 +172631,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -136636,8 +172674,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -136651,6 +172704,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -136677,7 +172731,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -136752,6 +172806,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -136784,6 +172848,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -136815,11 +172889,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<westmere::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<westmere::ondemand::value> value() noexcept;
};
@@ -136827,6 +172907,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for westmere */
+/* including simdjson/generic/ondemand/key_selector.h for westmere: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for westmere */
/* including simdjson/generic/ondemand/object.h for westmere: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for westmere */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -136836,6 +174308,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -136846,6 +174319,114 @@ namespace simdjson {
namespace westmere {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -136864,8 +174445,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -136877,10 +174469,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -136953,6 +174546,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -137029,6 +174716,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -137074,7 +174789,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -137086,7 +174801,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -137138,10 +174853,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -137157,7 +174880,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<westmere::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<westmere::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<westmere::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -137175,6 +174899,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<westmere::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(westmere::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -137182,7 +174908,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, westmere::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, westmere::ondemand::object>) {
return first;
@@ -137190,7 +174916,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, westmere::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, westmere::ondemand::object>) {
out = first;
@@ -137200,6 +174926,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires westmere::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, westmere::ondemand::value>
+ simdjson_inline westmere::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, westmere::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires westmere::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline westmere::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline westmere::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -137240,6 +174999,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -137259,6 +175027,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -137304,6 +175075,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for westmere */
+/* including simdjson/generic/ondemand/ranges.h for westmere: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::westmere::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::westmere::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for westmere */
/* including simdjson/generic/ondemand/serialization.h for westmere: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for westmere */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -137436,12 +175392,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -137466,10 +175424,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -137505,11 +175478,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -137533,22 +175554,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -137589,7 +175653,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -137608,21 +175672,21 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out)
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::value &val, T &out) noexcept(false) {
westmere::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::document &doc, T &out) noexcept(false) {
westmere::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &doc, T &out) noexcept(false) {
westmere::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -137633,10 +175697,6 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &d
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -137644,7 +175704,7 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &d
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -137656,12 +175716,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -137693,53 +175754,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, westmere::ondemand::number>
+&& !std::is_same_v<T, westmere::ondemand::document>
+&& !std::is_same_v<T, westmere::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^westmere::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = westmere::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, westmere::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ westmere::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ westmere::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ westmere::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ westmere::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, westmere::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
westmere::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, westmere::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, westmere::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, westmere::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -137755,33 +176358,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -138093,9 +176688,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -138122,6 +176725,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -138135,6 +176741,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -138142,31 +176751,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -138240,6 +176848,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -138269,10 +176880,14 @@ simdjson_inline simdjson_result<westmere::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -138335,6 +176950,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -138430,6 +177098,41 @@ namespace simdjson {
namespace westmere {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -138461,6 +177164,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -138474,6 +177184,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -138487,17 +177203,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -138509,12 +177245,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -138522,12 +177272,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -138696,6 +177460,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -138733,6 +177500,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -138846,10 +177617,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::val
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -138858,6 +177665,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::onde
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -138886,11 +177699,23 @@ template<> simdjson_inline error_code simdjson_result<westmere::ondemand::value>
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -139160,16 +177985,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -139177,9 +178008,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -139201,11 +178059,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -139213,17 +178085,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -139562,6 +178452,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::doc
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -139570,10 +178476,36 @@ simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::docu
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -139601,22 +178533,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<westmere::ondemand::docume
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<westmere::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<westmere::ondemand::document>(first).get<T>(out);
}
@@ -139685,27 +178641,27 @@ simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -139795,21 +178751,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -139821,11 +178794,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -139971,6 +178958,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::doc
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -139979,10 +178982,36 @@ simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::docu
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -140009,22 +179038,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<westmere::ondemand::docume
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<westmere::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<westmere::ondemand::document_reference>(first).get<T>(out);
}
@@ -140086,27 +179139,27 @@ simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operato
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator westmere::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator westmere::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -140172,6 +179225,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondema
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -140258,23 +179312,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -140283,6 +179334,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -140302,6 +179354,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -140382,13 +179437,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -140463,12 +179525,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -140492,10 +179611,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -140504,11 +179648,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -140516,14 +179668,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -140648,11 +179833,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -140674,6 +179867,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -140718,11 +179917,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::onde
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -140766,6 +179979,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -140777,6 +179993,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -140803,7 +180022,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -140875,7 +180095,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -140918,7 +180139,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -140935,6 +180156,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -141217,7 +180449,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -141556,6 +180788,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -141585,12 +180821,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -141600,6 +180845,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -141609,6 +180857,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -141644,6 +181036,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -141665,9 +181060,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -141676,7 +181079,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -141778,6 +181183,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -141785,9 +181193,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -141845,10 +181285,14 @@ simdjson_inline simdjson_result<westmere::ondemand::object>::simdjson_result(wes
simdjson_inline simdjson_result<westmere::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<westmere::ondemand::object>(error) {}
-simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -141902,11 +181346,55 @@ simdjson_inline error_code simdjson_result<westmere::ondemand::object>::for_each
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires westmere::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, westmere::ondemand::value>
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, westmere::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires westmere::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<westmere::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<westmere::ondemand::object_position> simdjson_result<westmere::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<westmere::ondemand::object>::revert_position(westmere::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<westmere::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -141950,6 +181438,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -142077,6 +181620,147 @@ simdjson_inline simdjson_result<westmere::ondemand::object_iterator> &simdjson_r
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for westmere */
+/* including simdjson/generic/ondemand/ranges-inl.h for westmere: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::westmere::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::westmere::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::westmere::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::westmere::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::westmere::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::westmere::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for westmere */
/* including simdjson/generic/ondemand/parser-inl.h for westmere: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for westmere */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -142108,7 +181792,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -142133,6 +181820,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -142149,6 +181837,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -142214,6 +181903,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -142221,8 +181938,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -142242,6 +181962,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -142649,6 +182414,27 @@ namespace simdjson {
namespace westmere {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -143036,6 +182822,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -143153,7 +182955,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -143164,6 +182966,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -143197,6 +183008,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -143299,7 +183120,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -143311,6 +183132,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -143420,6 +183250,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -143648,6 +183515,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -145036,10 +184918,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lsx
@@ -145575,7 +185453,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/lsx/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for lsx: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for lsx */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -145624,6 +185502,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace lsx
} // namespace simdjson
@@ -145656,6 +185541,9 @@ template <> struct is_builtin_deserializable<lsx::ondemand::object> : std::true_
template <> struct is_builtin_deserializable<lsx::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<lsx::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -145673,6 +185561,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = lsx::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = lsx::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = lsx::ondemand::array;
@@ -145887,6 +185779,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -146039,6 +185942,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -146057,6 +185962,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -146192,6 +186099,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -146260,9 +186176,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -146272,7 +186191,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -146287,7 +186206,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -146297,7 +186217,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -146325,7 +186245,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -146334,7 +186254,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -146412,6 +186333,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -146428,6 +186393,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -146455,6 +186467,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -146542,7 +186574,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -147007,9 +187039,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -147017,9 +187064,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -147350,6 +187407,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -147441,6 +187499,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -147465,6 +187526,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -148344,33 +188409,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -148498,6 +188617,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -148560,8 +188680,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -148692,7 +188823,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -148704,7 +188835,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -148760,6 +188891,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -148781,7 +188916,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<lsx::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -148801,7 +188937,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lsx::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lsx::ondemand::array>) {
return first;
@@ -148809,7 +188945,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lsx::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lsx::ondemand::array>) {
out = first;
@@ -148861,6 +188997,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -148903,6 +189048,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -149002,14 +189150,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -149047,6 +189195,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -149062,6 +189250,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -149075,6 +189310,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -149145,9 +189398,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -149156,7 +189412,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -149179,7 +189435,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -149191,7 +189447,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -149202,7 +189459,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -149215,7 +189472,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -149224,7 +189481,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -149233,7 +189491,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -149267,24 +189530,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -149294,7 +189557,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -149303,14 +189566,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -149805,9 +190068,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -149819,7 +190097,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -149832,7 +190110,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -149844,7 +190123,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -149855,7 +190135,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -149868,7 +190148,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -149877,7 +190157,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -149886,7 +190167,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -149899,12 +190185,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -149966,9 +190252,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -149977,11 +190278,31 @@ public:
simdjson_inline simdjson_result<lsx::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using lsx::implementation_simdjson_result_base<lsx::ondemand::document>::operator*;
@@ -149990,12 +190311,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator lsx::ondemand::array() & noexcept(false);
simdjson_inline operator lsx::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator lsx::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -150061,9 +190382,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -150072,22 +190408,42 @@ public:
simdjson_inline simdjson_result<lsx::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator lsx::ondemand::array() & noexcept(false);
simdjson_inline operator lsx::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator lsx::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -150255,10 +190611,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -150267,6 +190620,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -150326,7 +190682,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -150380,13 +190739,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -150420,8 +190782,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -150435,6 +190812,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -150461,7 +190839,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -150536,6 +190914,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -150568,6 +190956,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -150599,11 +190997,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<lsx::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<lsx::ondemand::value> value() noexcept;
};
@@ -150611,6 +191015,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for lsx */
+/* including simdjson/generic/ondemand/key_selector.h for lsx: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for lsx */
/* including simdjson/generic/ondemand/object.h for lsx: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for lsx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -150620,6 +192416,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -150630,6 +192427,114 @@ namespace simdjson {
namespace lsx {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -150648,8 +192553,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -150661,10 +192577,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -150737,6 +192654,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -150813,6 +192824,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -150858,7 +192897,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -150870,7 +192909,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -150922,10 +192961,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -150941,7 +192988,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<lsx::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<lsx::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<lsx::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -150959,6 +193007,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<lsx::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(lsx::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -150966,7 +193016,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lsx::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lsx::ondemand::object>) {
return first;
@@ -150974,7 +193024,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lsx::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lsx::ondemand::object>) {
out = first;
@@ -150984,6 +193034,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires lsx::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, lsx::ondemand::value>
+ simdjson_inline lsx::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lsx::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires lsx::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline lsx::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline lsx::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -151024,6 +193107,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -151043,6 +193135,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -151088,6 +193183,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for lsx */
+/* including simdjson/generic/ondemand/ranges.h for lsx: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::lsx::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::lsx::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for lsx */
/* including simdjson/generic/ondemand/serialization.h for lsx: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for lsx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -151220,12 +193500,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -151250,10 +193532,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -151289,11 +193586,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -151317,22 +193662,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -151373,7 +193761,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -151392,21 +193780,21 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexc
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::value &val, T &out) noexcept(false) {
lsx::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::document &doc, T &out) noexcept(false) {
lsx::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T &out) noexcept(false) {
lsx::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -151417,10 +193805,6 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -151428,7 +193812,7 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -151440,12 +193824,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -151477,53 +193862,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, lsx::ondemand::number>
+&& !std::is_same_v<T, lsx::ondemand::document>
+&& !std::is_same_v<T, lsx::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^lsx::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = lsx::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, lsx::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ lsx::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ lsx::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ lsx::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ lsx::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lsx::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
lsx::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lsx::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, lsx::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, lsx::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -151539,33 +194466,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -151877,9 +194796,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -151906,6 +194833,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -151919,6 +194849,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -151926,31 +194859,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -152024,6 +194956,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -152053,10 +194988,14 @@ simdjson_inline simdjson_result<lsx::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -152119,6 +195058,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -152214,6 +195206,41 @@ namespace simdjson {
namespace lsx {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -152245,6 +195272,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -152258,6 +195292,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -152271,17 +195311,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -152293,12 +195353,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -152306,12 +195380,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -152480,6 +195568,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -152517,6 +195608,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -152630,10 +195725,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::value>::
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -152642,6 +195773,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand:
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -152670,11 +195807,23 @@ template<> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -152944,16 +196093,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -152961,9 +196116,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -152985,11 +196167,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -152997,17 +196193,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -153346,6 +196560,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::document
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -153354,10 +196584,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document>
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -153385,22 +196641,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lsx::ondemand::document>::
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lsx::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lsx::ondemand::document>(first).get<T>(out);
}
@@ -153469,27 +196749,27 @@ simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -153579,21 +196859,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -153605,11 +196902,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -153755,6 +197066,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::document
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -153763,10 +197090,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document_
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -153793,22 +197146,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lsx::ondemand::document_re
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lsx::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lsx::ondemand::document_reference>(first).get<T>(out);
}
@@ -153870,27 +197247,27 @@ simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -153956,6 +197333,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::d
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -154042,23 +197420,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -154067,6 +197442,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -154086,6 +197462,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -154166,13 +197545,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -154247,12 +197633,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -154276,10 +197719,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -154288,11 +197756,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -154300,14 +197776,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -154432,11 +197941,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -154458,6 +197975,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -154502,11 +198025,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand:
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -154550,6 +198087,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -154561,6 +198101,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -154587,7 +198130,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -154659,7 +198203,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -154702,7 +198247,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -154719,6 +198264,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -155001,7 +198557,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -155340,6 +198896,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -155369,12 +198929,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -155384,6 +198953,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -155393,6 +198965,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -155428,6 +199144,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -155449,9 +199168,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -155460,7 +199187,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -155562,6 +199291,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -155569,9 +199301,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -155629,10 +199393,14 @@ simdjson_inline simdjson_result<lsx::ondemand::object>::simdjson_result(lsx::ond
simdjson_inline simdjson_result<lsx::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<lsx::ondemand::object>(error) {}
-simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -155686,11 +199454,55 @@ simdjson_inline error_code simdjson_result<lsx::ondemand::object>::for_each_at_p
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires lsx::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, lsx::ondemand::value>
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lsx::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires lsx::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<lsx::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<lsx::ondemand::object_position> simdjson_result<lsx::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<lsx::ondemand::object>::revert_position(lsx::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<lsx::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -155734,6 +199546,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -155861,6 +199728,147 @@ simdjson_inline simdjson_result<lsx::ondemand::object_iterator> &simdjson_result
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for lsx */
+/* including simdjson/generic/ondemand/ranges-inl.h for lsx: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::lsx::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::lsx::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::lsx::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::lsx::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::lsx::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::lsx::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for lsx */
/* including simdjson/generic/ondemand/parser-inl.h for lsx: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for lsx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -155892,7 +199900,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -155917,6 +199928,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -155933,6 +199945,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -155998,6 +200011,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -156005,8 +200046,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -156026,6 +200070,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -156433,6 +200522,27 @@ namespace simdjson {
namespace lsx {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -156820,6 +200930,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -156937,7 +201063,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -156948,6 +201074,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -156981,6 +201116,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -157083,7 +201228,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -157095,6 +201240,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -157204,6 +201358,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -157432,6 +201623,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -158825,10 +203031,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace lasx
@@ -159382,7 +203584,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
/* end file simdjson/lasx/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for lasx: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for lasx */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -159431,6 +203633,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace lasx
} // namespace simdjson
@@ -159463,6 +203672,9 @@ template <> struct is_builtin_deserializable<lasx::ondemand::object> : std::true
template <> struct is_builtin_deserializable<lasx::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<lasx::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -159480,6 +203692,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = lasx::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = lasx::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = lasx::ondemand::array;
@@ -159694,6 +203910,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -159846,6 +204073,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -159864,6 +204093,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -159999,6 +204230,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -160067,9 +204307,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -160079,7 +204322,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -160094,7 +204337,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -160104,7 +204348,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -160132,7 +204376,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -160141,7 +204385,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -160219,6 +204464,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -160235,6 +204524,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -160262,6 +204598,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -160349,7 +204705,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -160814,9 +205170,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -160824,9 +205195,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -161157,6 +205538,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -161248,6 +205630,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -161272,6 +205657,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -162151,33 +206540,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -162305,6 +206748,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -162367,8 +206811,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -162499,7 +206954,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -162511,7 +206966,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -162567,6 +207022,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -162588,7 +207047,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<lasx::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -162608,7 +207068,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lasx::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lasx::ondemand::array>) {
return first;
@@ -162616,7 +207076,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lasx::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lasx::ondemand::array>) {
out = first;
@@ -162668,6 +207128,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -162710,6 +207179,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -162809,14 +207281,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -162854,6 +207326,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -162869,6 +207381,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -162882,6 +207441,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -162952,9 +207529,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -162963,7 +207543,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -162986,7 +207566,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -162998,7 +207578,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -163009,7 +207590,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -163022,7 +207603,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -163031,7 +207612,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -163040,7 +207622,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -163074,24 +207661,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -163101,7 +207688,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -163110,14 +207697,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -163612,9 +208199,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163626,7 +208228,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -163639,7 +208241,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -163651,7 +208254,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -163662,7 +208266,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -163675,7 +208279,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -163684,7 +208288,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -163693,7 +208298,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -163706,12 +208316,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -163773,9 +208383,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163784,11 +208409,31 @@ public:
simdjson_inline simdjson_result<lasx::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using lasx::implementation_simdjson_result_base<lasx::ondemand::document>::operator*;
@@ -163797,12 +208442,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator lasx::ondemand::array() & noexcept(false);
simdjson_inline operator lasx::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator lasx::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -163868,9 +208513,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163879,22 +208539,42 @@ public:
simdjson_inline simdjson_result<lasx::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator lasx::ondemand::array() & noexcept(false);
simdjson_inline operator lasx::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator lasx::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -164062,10 +208742,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -164074,6 +208751,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -164133,7 +208813,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -164187,13 +208870,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -164227,8 +208913,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -164242,6 +208943,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -164268,7 +208970,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -164343,6 +209045,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -164375,6 +209087,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -164406,11 +209128,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<lasx::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<lasx::ondemand::value> value() noexcept;
};
@@ -164418,6 +209146,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for lasx */
+/* including simdjson/generic/ondemand/key_selector.h for lasx: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for lasx */
/* including simdjson/generic/ondemand/object.h for lasx: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for lasx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -164427,6 +210547,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -164437,6 +210558,114 @@ namespace simdjson {
namespace lasx {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -164455,8 +210684,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -164468,10 +210708,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -164544,6 +210785,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -164620,6 +210955,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -164665,7 +211028,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -164677,7 +211040,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -164729,10 +211092,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -164748,7 +211119,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<lasx::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<lasx::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<lasx::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -164766,6 +211138,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<lasx::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(lasx::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -164773,7 +211147,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lasx::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lasx::ondemand::object>) {
return first;
@@ -164781,7 +211155,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lasx::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, lasx::ondemand::object>) {
out = first;
@@ -164791,6 +211165,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires lasx::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, lasx::ondemand::value>
+ simdjson_inline lasx::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lasx::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires lasx::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline lasx::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline lasx::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -164831,6 +211238,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -164850,6 +211266,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -164895,6 +211314,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for lasx */
+/* including simdjson/generic/ondemand/ranges.h for lasx: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::lasx::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::lasx::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for lasx */
/* including simdjson/generic/ondemand/serialization.h for lasx: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for lasx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -165027,12 +211631,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -165057,10 +211663,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -165096,11 +211717,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -165124,22 +211793,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -165180,7 +211892,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -165199,21 +211911,21 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noex
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::value &val, T &out) noexcept(false) {
lasx::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::document &doc, T &out) noexcept(false) {
lasx::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc, T &out) noexcept(false) {
lasx::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -165224,10 +211936,6 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc,
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -165235,7 +211943,7 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc,
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -165247,12 +211955,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -165284,53 +211993,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, lasx::ondemand::number>
+&& !std::is_same_v<T, lasx::ondemand::document>
+&& !std::is_same_v<T, lasx::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^lasx::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = lasx::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, lasx::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ lasx::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ lasx::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ lasx::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ lasx::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lasx::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
lasx::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lasx::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, lasx::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, lasx::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -165346,33 +212597,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -165684,9 +212927,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -165713,6 +212964,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -165726,6 +212980,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -165733,31 +212990,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -165831,6 +213087,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -165860,10 +213119,14 @@ simdjson_inline simdjson_result<lasx::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -165926,6 +213189,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -166021,6 +213337,41 @@ namespace simdjson {
namespace lasx {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -166052,6 +213403,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -166065,6 +213423,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -166078,17 +213442,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -166100,12 +213484,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -166113,12 +213511,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -166287,6 +213699,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -166324,6 +213739,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -166437,10 +213856,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::value>:
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -166449,6 +213904,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -166477,11 +213938,23 @@ template<> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::ge
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -166751,16 +214224,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -166768,9 +214247,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -166792,11 +214298,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -166804,17 +214324,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -167153,6 +214691,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::documen
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -167161,10 +214715,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -167192,22 +214772,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lasx::ondemand::document>:
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lasx::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lasx::ondemand::document>(first).get<T>(out);
}
@@ -167276,27 +214880,27 @@ simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondema
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -167386,21 +214990,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -167412,11 +215033,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -167562,6 +215197,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::documen
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -167570,10 +215221,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -167600,22 +215277,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lasx::ondemand::document_r
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lasx::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<lasx::ondemand::document_reference>(first).get<T>(out);
}
@@ -167677,27 +215378,27 @@ simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator la
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator lasx::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator lasx::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -167763,6 +215464,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -167849,23 +215551,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -167874,6 +215573,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -167893,6 +215593,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -167973,13 +215676,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -168054,12 +215764,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -168083,10 +215850,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -168095,11 +215887,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -168107,14 +215907,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -168239,11 +216072,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -168265,6 +216106,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -168309,11 +216156,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -168357,6 +216218,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -168368,6 +216232,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -168394,7 +216261,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -168466,7 +216334,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -168509,7 +216378,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -168526,6 +216395,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -168808,7 +216688,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -169147,6 +217027,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -169176,12 +217060,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -169191,6 +217084,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -169200,6 +217096,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -169235,6 +217275,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -169256,9 +217299,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -169267,7 +217318,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -169369,6 +217422,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -169376,9 +217432,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -169436,10 +217524,14 @@ simdjson_inline simdjson_result<lasx::ondemand::object>::simdjson_result(lasx::o
simdjson_inline simdjson_result<lasx::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<lasx::ondemand::object>(error) {}
-simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -169493,11 +217585,55 @@ simdjson_inline error_code simdjson_result<lasx::ondemand::object>::for_each_at_
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires lasx::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, lasx::ondemand::value>
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lasx::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires lasx::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<lasx::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<lasx::ondemand::object_position> simdjson_result<lasx::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<lasx::ondemand::object>::revert_position(lasx::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<lasx::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -169541,6 +217677,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -169668,6 +217859,147 @@ simdjson_inline simdjson_result<lasx::ondemand::object_iterator> &simdjson_resul
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for lasx */
+/* including simdjson/generic/ondemand/ranges-inl.h for lasx: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::lasx::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::lasx::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::lasx::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::lasx::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::lasx::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::lasx::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for lasx */
/* including simdjson/generic/ondemand/parser-inl.h for lasx: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for lasx */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -169699,7 +218031,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -169724,6 +218059,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -169740,6 +218076,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -169805,6 +218142,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -169812,8 +218177,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -169833,6 +218201,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -170240,6 +218653,27 @@ namespace simdjson {
namespace lasx {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -170627,6 +219061,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -170744,7 +219194,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -170755,6 +219205,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -170788,6 +219247,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -170890,7 +219359,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -170902,6 +219371,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -171011,6 +219489,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -171239,6 +219754,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -172641,11 +221171,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
return __builtin_popcountll(input_num);
}
-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
- uint64_t *result) {
- return __builtin_uaddll_overflow(value1, value2,
- reinterpret_cast<unsigned long long *>(result));
-}
} // unnamed namespace
} // namespace rvv_vls
@@ -173193,7 +221718,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
/* end file simdjson/rvv-vls/begin.h */
/* including simdjson/generic/ondemand/amalgamated.h for rvv_vls: #include "simdjson/generic/ondemand/amalgamated.h" */
/* begin file simdjson/generic/ondemand/amalgamated.h for rvv_vls */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
#endif
@@ -173242,6 +221767,13 @@ class token_iterator;
class value;
class value_iterator;
+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
} // namespace ondemand
} // namespace rvv_vls
} // namespace simdjson
@@ -173274,6 +221806,9 @@ template <> struct is_builtin_deserializable<rvv_vls::ondemand::object> : std::t
template <> struct is_builtin_deserializable<rvv_vls::ondemand::value> : std::true_type {};
template <> struct is_builtin_deserializable<rvv_vls::ondemand::raw_json_string> : std::true_type {};
template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename T>
concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -173291,6 +221826,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
template <typename T, typename ValT = rvv_vls::ondemand::value>
concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;
+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = rvv_vls::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
/// Deserialize Tag
inline constexpr struct deserialize_tag {
using array_type = rvv_vls::ondemand::array;
@@ -173505,6 +222044,17 @@ public:
*/
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;
+ /**
+ * Get the current field's key together with its raw byte length.
+ *
+ * Like field_key(), but also returns the number of raw key bytes (the distance
+ * from the first key byte to the closing quote). The length is recovered from
+ * the structural index -- the next structural token is the ':' -- by stepping
+ * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+ * for the closing quote. Leaves the iterator positioned exactly as field_key().
+ */
+ simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
/**
* Pass the : in the field and move to its value.
*/
@@ -173657,6 +222207,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -173675,6 +222227,8 @@ public:
simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+ simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -173810,6 +222364,15 @@ protected:
/** @copydoc error_code json_iterator::position() const noexcept; */
simdjson_inline token_position position() const noexcept;
+ /**
+ * Move the live iterator directly to the given position and depth, without
+ * validating against the parser's per-depth container-start bookkeeping
+ * (unlike json_iterator::reenter_child()). Used to restore a previously
+ * captured mid-container position (see object::revert_position()): that
+ * bookkeeping only tracks each container's own start, not every position
+ * a caller might later capture and revert to, so it does not apply here.
+ */
+ simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
simdjson_inline token_position last_position() const noexcept;
/** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -173878,9 +222441,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
* When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
*
@@ -173890,7 +222456,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get()
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -173905,7 +222471,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
* If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
*
* @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -173915,7 +222482,7 @@ public:
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, value>)
#else
noexcept
#endif
@@ -173943,7 +222510,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -173952,7 +222519,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -174030,6 +222598,50 @@ public:
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
/**
* Cast this JSON value to a double.
*
@@ -174046,6 +222658,53 @@ public:
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
/**
* Cast this JSON value to a string.
*
@@ -174073,6 +222732,26 @@ public:
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: a value should be consumed once. Calling get_u8string() twice on the same
+ * value is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -174160,7 +222839,7 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline operator uint64_t() noexcept(false);
@@ -174625,9 +223304,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -174635,9 +223329,19 @@ public:
simdjson_inline simdjson_result<bool> get_bool() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) noexcept;
+ template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
@@ -174968,6 +223672,7 @@ protected:
token_position _position{};
friend class json_iterator;
+ friend class document_stream;
friend class value_iterator;
friend class object;
template <typename... Args>
@@ -175059,6 +223764,9 @@ protected:
* value of this attribute.
*/
bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ bool _allow_incomplete_json{false};
+#endif
public:
simdjson_inline json_iterator() noexcept = default;
@@ -175083,6 +223791,10 @@ public:
* start_root_array() and start_root_object().
*/
simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ simdjson_inline bool allow_incomplete_json() const noexcept;
+ simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
/**
* Get the root value iterator
@@ -175962,33 +224674,87 @@ public:
* @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
* spot is cache-related: small enough to fit in cache, yet big enough to
* parse as many documents as possible in one tight loop.
- * Defaults to 10MB, which has been a reasonable sweet spot in our tests.
- * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
- * separated by commas instead of whitespace. It comes with a performance
- * penalty because the entire document is indexed at once (and the document must be
- * less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
- * is effectively ignored, as it is set to at least the document size.
+ * Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+ * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+ * When true, maps internally to stream_format::comma_delimited.
+ * Defaults to false.
* @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
* - MEMALLOC if the parser does not have enough capacity and memory allocation fails
* - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
* - other json errors if parsing fails. You should not rely on these errors to always the same for the
* same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
*/
- inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
- inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
- /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
- inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+ /** @private An rvalue input is destroyed at the end of the full-expression, while the
+ * returned document_stream only holds a pointer to it: iterating the stream would then
+ * read freed memory. These deleted overloads also catch a std::string_view argument,
+ * which would otherwise convert implicitly to a padded_string temporary. */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
/** @private We do not want to allow implicit conversion from C string to std::string. */
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+ /**
+ * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+ */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+ simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+ /**
+ * Parse a stream of JSON documents with explicit format specification.
+ *
+ * @param buf The concatenated JSON documents.
+ * @param len The length of the buffer.
+ * @param batch_size The batch size to use.
+ * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+ * @return A stream of documents, or an error.
+ */
+ inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+ *
+ * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+ */
+ inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+ /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+ /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+ inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+ /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+ inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
/** The capacity of this parser (the largest document it can process). */
simdjson_pure simdjson_inline size_t capacity() const noexcept;
/** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -176116,6 +224882,7 @@ private:
size_t _capacity{0};
size_t _max_capacity;
size_t _max_depth{DEFAULT_MAX_DEPTH};
+ size_t _document_len{0};
std::unique_ptr<uint8_t[]> string_buf{};
#if SIMDJSON_DEVELOPMENT_CHECKS
@@ -176178,8 +224945,19 @@ public:
* Begin array iteration.
*
* Part of the std::iterable interface.
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+ * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+ * reported as OUT_OF_ORDER_ITERATION.
*/
- simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+ /**
+ * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+ *
+ * The iterator does not depend on the array instance and may outlive it, so
+ * it does not lock it.
+ */
+ simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
/**
* Sentinel representing the end of the array.
*
@@ -176310,7 +225088,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+ noexcept(nothrow_gettable<T, array>) {
static_assert(custom_deserializable<T, array>);
return deserialize(*this, out);
}
@@ -176322,7 +225100,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, array>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -176378,6 +225156,10 @@ protected:
* iter.is_alive() == false indicates iteration is complete.
*/
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
@@ -176399,7 +225181,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() && noexcept;
simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> end() noexcept;
inline simdjson_result<size_t> count_elements() & noexcept;
inline simdjson_result<bool> is_empty() & noexcept;
@@ -176419,7 +225202,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, rvv_vls::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, rvv_vls::ondemand::array>) {
return first;
@@ -176427,7 +225210,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, rvv_vls::ondemand::array>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, rvv_vls::ondemand::array>) {
out = first;
@@ -176479,6 +225262,15 @@ public:
/** Create a new, invalid array iterator. */
simdjson_inline array_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~array_iterator() noexcept;
+
+ simdjson_inline array_iterator(array_iterator&&) noexcept;
+ simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+ simdjson_inline array_iterator(const array_iterator&) noexcept;
+ simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -176521,6 +225313,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ array* parent{nullptr};
+
+ simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
#endif
value_iterator iter{};
@@ -176620,14 +225415,14 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
/**
* Cast this JSON value (inside string) to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
*/
simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -176665,6 +225460,46 @@ public:
* @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
*/
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint16_t.
+ *
+ * @returns A 16-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+ */
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ /**
+ * Cast this JSON value to a 16-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int16_t.
+ *
+ * @returns A 16-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+ */
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit unsigned integer.
+ *
+ * Calls get_uint64() and checks that the result fits in a uint8_t.
+ *
+ * @returns An 8-bit unsigned integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+ */
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ /**
+ * Cast this JSON value to an 8-bit signed integer.
+ *
+ * Calls get_int64() and checks that the result fits in an int8_t.
+ *
+ * @returns An 8-bit signed integer.
+ * @returns INCORRECT_TYPE If the JSON value is not an integer.
+ * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+ */
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
/**
* Cast this JSON value to a double.
*
@@ -176680,6 +225515,53 @@ public:
* @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
*/
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+ /**
+ * Cast this JSON value to a float (binary32).
+ *
+ * The value is rounded directly to binary32: it is the float nearest to the
+ * JSON number. Note that this may differ from get_double() followed by a cast
+ * to float, which rounds twice.
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+
+ /**
+ * Cast this JSON value (inside string) to a float (binary32).
+ *
+ * @returns A float.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ /**
+ * Cast this JSON value to a std::float32_t (C++23).
+ *
+ * Same as get_float(): the value is rounded directly to binary32.
+ *
+ * @returns A std::float32_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+ */
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ /**
+ * Cast this JSON value to a std::float64_t (C++23).
+ *
+ * Same as get_double().
+ *
+ * @returns A std::float64_t.
+ * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+ */
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
/**
* Cast this JSON value to a string.
*
@@ -176693,6 +225575,24 @@ public:
* @returns INCORRECT_TYPE if the JSON value is not a string.
*/
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Cast this JSON value to a C++20 UTF-8 string.
+ *
+ * The string is guaranteed to be valid UTF-8.
+ *
+ * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+ * get_string(): the very same bytes are returned, viewed as char8_t.
+ *
+ * Important: Calling get_u8string() twice on the same document is an error.
+ *
+ * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+ * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+ * time it parses a document or when it is destroyed.
+ * @returns INCORRECT_TYPE if the JSON value is not a string.
+ */
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Attempts to fill the provided std::string reference with the parsed value of the current string.
*
@@ -176763,9 +225663,12 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool
*
- * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+ * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+ * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+ * get_float32(), get_float64(),
* get_object(), get_array(), get_raw_json_string(), or get_string() instead.
*
* @returns A value of the given type, parsed from the JSON.
@@ -176774,7 +225677,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -176797,7 +225700,7 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -176809,7 +225712,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -176820,7 +225724,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -176833,7 +225737,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -176842,7 +225746,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -176851,7 +225756,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
/**
@@ -176885,24 +225795,24 @@ public:
/**
* Cast this JSON value to an unsigned integer.
*
- * @returns A signed 64-bit integer.
+ * @returns A unsigned 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
*/
- simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
/**
* Cast this JSON value to a signed integer.
*
* @returns A signed 64-bit integer.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
*/
- simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
/**
* Cast this JSON value to a double.
*
* @returns A double.
* @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
*/
- simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
/**
* Cast this JSON value to a string.
*
@@ -176912,7 +225822,7 @@ public:
* time it parses a document or when it is destroyed.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a raw_json_string.
*
@@ -176921,14 +225831,14 @@ public:
* @returns A pointer to the raw JSON for the given string.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
*/
- simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+ explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
/**
* Cast this JSON value to a bool.
*
* @returns A bool value.
* @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
*/
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
/**
* Cast this JSON value to a value when the document is an object or an array.
*
@@ -177423,9 +226333,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177437,7 +226362,7 @@ public:
template <typename T>
simdjson_inline simdjson_result<T> get() &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -177450,7 +226375,8 @@ public:
template<typename T>
simdjson_inline simdjson_result<T> get() &&
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+ // Forwards to document::get<T>(), so the document customization decides.
+ noexcept(nothrow_gettable<T, document>)
#else
noexcept
#endif
@@ -177462,7 +226388,8 @@ public:
/**
* Get this value as the given type.
*
- * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+ * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+ * std::float64_t and std::float32_t (C++23, when available), bool, value
*
* Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
*
@@ -177473,7 +226400,7 @@ public:
template<typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out) &
#if SIMDJSON_SUPPORTS_CONCEPTS
- noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+ noexcept(nothrow_gettable<T, document_reference>)
#else
noexcept
#endif
@@ -177486,7 +226413,7 @@ public:
"And you do not seem to have added support for it. Indeed, we have that "
"simdjson::custom_deserializable<T> is false and the type T is not a default type "
"such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, or bool.");
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
static_cast<void>(out); // to get rid of unused errors
return UNINITIALIZED;
}
@@ -177495,7 +226422,8 @@ public:
// immediately fail.
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
"The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
- "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+ "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+ "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
" get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
" You may also add support for custom types, see our documentation.");
static_cast<void>(out); // to get rid of unused errors
@@ -177504,7 +226432,12 @@ public:
}
/** @overload template<typename T> error_code get(T &out) & noexcept */
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, document_reference>);
+#else
+ noexcept;
+#endif
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
#if SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -177517,12 +226450,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator array() & noexcept(false);
simdjson_inline operator object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177584,9 +226517,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177595,11 +226543,31 @@ public:
simdjson_inline simdjson_result<rvv_vls::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
using rvv_vls::implementation_simdjson_result_base<rvv_vls::ondemand::document>::operator*;
@@ -177608,12 +226576,12 @@ public:
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator rvv_vls::ondemand::array() & noexcept(false);
simdjson_inline operator rvv_vls::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator rvv_vls::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177679,9 +226647,24 @@ public:
simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+ simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+ simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+ simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+ simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
simdjson_inline simdjson_result<double> get_double() noexcept;
simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+ simdjson_inline simdjson_result<float> get_float() noexcept;
+ simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+ simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177690,22 +226673,42 @@ public:
simdjson_inline simdjson_result<rvv_vls::ondemand::value> get_value() noexcept;
simdjson_inline simdjson_result<bool> is_null() noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
- template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+ template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
- template<typename T> simdjson_inline error_code get(T &out) & noexcept;
- template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+ template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
+ template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+ noexcept;
+#endif
#if SIMDJSON_EXCEPTIONS
template <class T>
explicit simdjson_inline operator T() noexcept(false);
simdjson_inline operator rvv_vls::ondemand::array() & noexcept(false);
simdjson_inline operator rvv_vls::ondemand::object() & noexcept(false);
- simdjson_inline operator uint64_t() noexcept(false);
- simdjson_inline operator int64_t() noexcept(false);
- simdjson_inline operator double() noexcept(false);
- simdjson_inline operator std::string_view() noexcept(false);
- simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
- simdjson_inline operator bool() noexcept(false);
+ explicit simdjson_inline operator uint64_t() noexcept(false);
+ explicit simdjson_inline operator int64_t() noexcept(false);
+ explicit simdjson_inline operator double() noexcept(false);
+ explicit simdjson_inline operator std::string_view() noexcept(false);
+ explicit simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
+ explicit simdjson_inline operator bool() noexcept(false);
simdjson_inline operator rvv_vls::ondemand::value() noexcept(false);
#endif
simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177873,10 +226876,7 @@ public:
* }
* size_t truncated = stream.truncated_bytes();
*
- * IMPORTANT: this value is only meaningful under the conditions below. It is
- * computed from stage-1 bookkeeping, and outside these conditions it is not
- * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
- * around to a huge value. Check it only when both of the following hold:
+ * IMPORTANT: this value is only meaningful under the conditions below.
*
* - you iterated all the way to the end of the stream;
* - no document reported an error. Iteration stops at the first failed
@@ -177885,6 +226885,9 @@ public:
* If you need to know about a truncated tail outside those conditions, track
* it yourself from the last successful document (see iterator::current_index()
* and iterator::source()).
+ *
+ * An empty input (zero bytes) or an input made only of white space contains
+ * no document: truncated_bytes() returns zero.
*/
inline size_t truncated_bytes() const noexcept;
@@ -177944,7 +226947,10 @@ public:
*
* The returned string_view instance is simply a map to the (unparsed)
* source string: it may thus include white-space characters and all manner
- * of padding.
+ * of padding. It spans the whole current document, whether or not you
+ * have already accessed (part of) the document. Thus
+ * current_index() + source().size() is the offset just past the end of the
+ * current document, which is useful when reading a stream in chunks.
*
* This function (source()) is experimental and the usage
* may change in future versions of simdjson: we find the API somewhat
@@ -177998,13 +227004,16 @@ private:
* @param buf is the raw byte buffer we need to process
* @param len is the length of the raw byte buffer in bytes
* @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+ * @param allow_comma_separated whether to allow comma-separated documents
+ * @param format the stream format
*/
simdjson_inline document_stream(
ondemand::parser &parser,
const uint8_t *buf,
size_t len,
size_t batch_size,
- bool allow_comma_separated
+ bool allow_comma_separated,
+ stream_format format = stream_format::whitespace_delimited
) noexcept;
/**
@@ -178038,8 +227047,23 @@ private:
*/
inline void next() noexcept;
- /** Move the json_iterator of the document to the location of the next document in the stream. */
+ /**
+ * Move the json_iterator of the document to the location of the next document
+ * in the stream.
+ *
+ * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+ * when the iterator is still inside the current document (`depth() > 0`), this
+ * may jump to the next delimiter instead of walking remaining structurals. That
+ * jump does not structure-validate the unread remainder.
+ */
inline void next_document() noexcept;
+ /** Byte that ends a document under `format`, or 0 if there is none. */
+ simdjson_inline uint8_t document_delimiter() const noexcept;
+ /**
+ * Position the iterator at the first structural at or past the next
+ * `delimiter` in the current batch. Returns false if none is found.
+ */
+ simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;
/** Get the next document index. */
inline size_t next_batch_start() const noexcept;
@@ -178053,6 +227077,7 @@ private:
size_t len;
size_t batch_size;
bool allow_comma_separated;
+ stream_format format;
/**
* We are going to use just one document instance. The document owns
* the json_iterator. It implies that we only ever pass a reference
@@ -178079,7 +227104,7 @@ private:
/** The error returned from the stage 1 thread. */
error_code stage1_thread_error{UNINITIALIZED};
/** The thread used to run stage 1 against the next batch in the background. */
- std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+ std::unique_ptr<stage1_worker> worker{};
/**
* The parser used to run stage 1 in the background. Will be swapped
* with the regular parser when finished.
@@ -178154,6 +227179,16 @@ public:
* call it again nor can you call key().
*/
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+ *
+ * This consumes the key: once you have called unescaped_u8key(), you cannot
+ * call it again nor can you call key().
+ */
+ simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the key as a string_view (for higher speed, consider raw_key).
* We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -178186,6 +227221,16 @@ public:
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
*/
simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ /**
+ * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+ * escaped_key(): the very same bytes are returned, viewed as char8_t.
+ * The string is unprocessed, so it may contain escape characters
+ * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+ * you can safely call it repeatedly.
+ */
+ simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
/**
* Get the field value.
*/
@@ -178217,11 +227262,17 @@ public:
simdjson_inline simdjson_result() noexcept = default;
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<typename string_type>
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
simdjson_inline simdjson_result<rvv_vls::ondemand::raw_json_string> key() noexcept;
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+ simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
simdjson_inline simdjson_result<rvv_vls::ondemand::value> value() noexcept;
};
@@ -178229,6 +227280,1398 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
/* end file simdjson/generic/ondemand/field.h for rvv_vls */
+/* including simdjson/generic/ondemand/key_selector.h for rvv_vls: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string> // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring> // std::memcpy (portable unaligned window load)
+#include <utility> // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+ #include <arm_neon.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+ #include <emmintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+ #include <lsxintrin.h>
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+ #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+ if (n == 0) { return 1; }
+ std::size_t p = 1;
+ while (p < n) { p <<= 1; }
+ return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+ if (pos == LAST_CHAR) {
+ if (key.empty()) { return 256; }
+ return static_cast<unsigned char>(key[key.size() - 1]);
+ }
+ if (pos >= key.size()) { return 256; }
+ return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+ bool distinguished = false;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+ distinguished = true;
+ break;
+ }
+ }
+ if (!distinguished) { ++count; }
+ }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* positions,
+ std::size_t num_positions,
+ std::size_t modulus) {
+ return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+ const std::array<std::string_view, N>& keys,
+ std::size_t pos,
+ std::size_t modulus) {
+ struct pair { std::size_t len_mod; std::size_t ch; };
+ std::array<pair, N> pairs{};
+ for (std::size_t i = 0; i < N; ++i) {
+ pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+ }
+ std::size_t count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ bool dup = false;
+ for (std::size_t j = 0; j < i; ++j) {
+ if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+ dup = true;
+ break;
+ }
+ }
+ if (!dup) { ++count; }
+ }
+ return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].size() > m) { m = keys[i].size(); }
+ }
+ return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+ const std::array<std::string_view, N>& keys,
+ const std::size_t* candidates,
+ std::size_t num_candidates,
+ std::size_t* positions,
+ std::size_t& num_positions_out,
+ std::size_t& budget,
+ std::size_t modulus) {
+ constexpr std::size_t MAX_DEPTH = 8;
+ std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+ struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+ std::array<frame, MAX_DEPTH + 1> stack{};
+ std::size_t sp = 0;
+
+ std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+ if (budget > 0) { --budget; }
+ if (initial_count == 0) { num_positions_out = 0; return true; }
+
+ stack[0] = {0, 0, initial_count};
+
+ while (budget > 0) {
+ if (sp > MAX_DEPTH) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ auto& f = stack[sp];
+ if (f.next_ci >= breadth) {
+ if (sp == 0) { break; }
+ --sp;
+ ++stack[sp].next_ci;
+ continue;
+ }
+ positions[sp] = candidates[f.next_ci];
+ --budget;
+ std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+ if (new_count == 0) { num_positions_out = sp + 1; return true; }
+ if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+ stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+ ++sp;
+ } else {
+ ++f.next_ci;
+ }
+ }
+ return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::size_t modulus) {
+ if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+ std::size_t max_len = max_key_length(keys);
+ constexpr std::size_t MAX_CANDIDATES = 256;
+ std::array<std::size_t, MAX_CANDIDATES> candidates{};
+ std::array<std::size_t, MAX_CANDIDATES> powers{};
+ std::size_t num_candidates = 0;
+ for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+ candidates[num_candidates] = p;
+ powers[num_candidates] = discriminating_power(keys, p, modulus);
+ ++num_candidates;
+ }
+ if (num_candidates < MAX_CANDIDATES) {
+ candidates[num_candidates] = LAST_CHAR;
+ powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+ ++num_candidates;
+ }
+ for (std::size_t i = 0; i < num_candidates; ++i) {
+ for (std::size_t j = i + 1; j < num_candidates; ++j) {
+ if (powers[j] > powers[i]) {
+ auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+ auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+ }
+ }
+ }
+
+ positions[0] = candidates[0];
+ if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+ if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+ {
+ std::size_t budget = 5000;
+ std::size_t num_found = 0;
+ if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+ positions.data(), num_found, budget, modulus)) {
+ return num_found;
+ }
+ }
+
+ std::size_t num_pos = 0;
+ for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+ bool already = false;
+ for (std::size_t p = 0; p < num_pos; ++p) {
+ if (positions[p] == candidates[ci]) { already = true; break; }
+ }
+ if (already) { continue; }
+ positions[num_pos] = candidates[ci];
+ ++num_pos;
+ if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+ }
+
+ compile_time_error("Failed to find distinguishing positions for perfect hash");
+ return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+ // Allow up to 8x the minimum table size. Sparser tables solve faster.
+ static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+ std::size_t table_size{};
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+ std::size_t num_positions{};
+ std::array<std::size_t, MAX_POSITIONS> positions{};
+ std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+ }
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+ for (std::size_t k = 0; k < N; ++k) {
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ kchars[k][p] = char_at(keys[k], positions[p]);
+ }
+ }
+
+ struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+ constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+ std::array<sym_t, MAX_SYMS> syms{};
+ std::size_t nsyms = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::array<std::size_t, 256> freq{};
+ for (std::size_t k = 0; k < N; ++k) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { freq[c]++; }
+ }
+ for (std::size_t c = 0; c < 256; ++c) {
+ if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+ }
+ }
+ for (std::size_t i = 0; i < nsyms; ++i) {
+ for (std::size_t j = i + 1; j < nsyms; ++j) {
+ if (syms[j].freq > syms[i].freq) {
+ auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+ }
+ }
+ }
+
+ std::array<std::size_t, N> phash{};
+ for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+ std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+ {
+ uint64_t s = 0x9e3779b97f4a7c15ULL;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+ salt[p][c] = s;
+ }
+ }
+ }
+ std::array<uint64_t, N> sig{};
+ for (std::size_t k = 0; k < N; ++k) {
+ uint64_t s = 0;
+ for (std::size_t p = 0; p < num_positions; ++p) {
+ std::size_t c = kchars[k][p];
+ if (c < 256) { s ^= salt[p][c]; }
+ }
+ sig[k] = s;
+ }
+ std::array<std::size_t, N> order{};
+ for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+ std::array<std::size_t, M> slot_gen{};
+ std::size_t gen = 0;
+
+ std::size_t search_limit = next_power_of_2(M);
+ if (search_limit < 32) { search_limit = 32; }
+
+ for (std::size_t si = 0; si < nsyms; ++si) {
+ std::size_t sp = syms[si].pos;
+ std::size_t sc = syms[si].ch;
+
+ uint64_t sp_salt = salt[sp][sc];
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+ }
+
+ for (std::size_t i = 1; i < N; ++i) {
+ std::size_t x = order[i];
+ uint64_t xs = sig[x];
+ std::size_t j = i;
+ while (j > 0 && sig[order[j - 1]] > xs) {
+ order[j] = order[j - 1];
+ --j;
+ }
+ order[j] = x;
+ }
+
+ bool found = false;
+ for (std::size_t v = 0; v < search_limit && !found; ++v) {
+ bool collision = false;
+ std::size_t ci = 0;
+ while (ci < N && !collision) {
+ uint64_t class_sig = sig[order[ci]];
+ std::size_t cj = ci;
+ while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+ if (cj - ci > 1) {
+ ++gen;
+ for (std::size_t x = ci; x < cj; ++x) {
+ std::size_t k = order[x];
+ std::size_t h = phash[k];
+ if (kchars[k][sp] == sc) { h += v; }
+ h %= M;
+ if (slot_gen[h] == gen) { collision = true; break; }
+ slot_gen[h] = gen;
+ }
+ }
+ ci = cj;
+ }
+ if (!collision) {
+ asso_values[sp][sc] = v;
+ for (std::size_t k = 0; k < N; ++k) {
+ if (kchars[k][sp] == sc) { phash[k] += v; }
+ }
+ found = true;
+ }
+ }
+ if (!found) { return false; }
+ }
+
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = phash[i] % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ if (try_compute_phf<N, M>(keys, result)) { return true; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+ return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+ std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+ std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+ return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+ std::size_t has = static_cast<std::size_t>(idx < len);
+ std::size_t si = idx & (std::size_t{0} - has);
+ return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+ std::size_t kc = key.size();
+ kc = kc * 31 + static_cast<unsigned char>(key[0]);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+ kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+ return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+ const std::array<std::string_view, N>& keys,
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+ std::size_t& num_positions,
+ std::array<std::size_t, MAX_POSITIONS>& positions,
+ std::array<std::size_t, M>& slot_to_key) {
+ num_positions = select_positions<N>(keys, positions, M);
+
+ if (num_positions == 0) {
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t slot = keys[i].size() % M;
+ if (slot_to_key[slot] != N) { return false; }
+ slot_to_key[slot] = i;
+ }
+ return true;
+ }
+
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ num_positions = HD_MODE; // sentinel for H&D mode
+ positions[0] = 0;
+ positions[1] = LAST_CHAR;
+
+ std::array<std::size_t, N> key_bucket{};
+ for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+ struct bucket_info { std::size_t ch; std::size_t count; };
+ std::array<bucket_info, N> buckets{};
+ std::size_t num_buckets = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ std::size_t bk = key_bucket[i];
+ bool found = false;
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+ }
+ if (!found) { buckets[num_buckets++] = {bk, 1}; }
+ }
+ for (std::size_t i = 0; i < num_buckets; ++i) {
+ for (std::size_t j = i + 1; j < num_buckets; ++j) {
+ if (buckets[j].count > buckets[i].count) {
+ auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+ }
+ }
+ }
+
+ auto try_placement = [&](auto key_hash_fn) -> bool {
+ for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+ for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+ for (std::size_t b = 0; b < num_buckets; ++b) {
+ std::size_t ch = buckets[b].ch;
+ std::array<std::size_t, N> bucket_keys{};
+ std::size_t bk_count = 0;
+ for (std::size_t i = 0; i < N; ++i) {
+ if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+ }
+ bool placed = false;
+ std::size_t max_d = M < 255 ? M : 255;
+ for (std::size_t d = 0; d < max_d; ++d) {
+ bool ok = true;
+ std::array<std::size_t, N> bucket_slots{};
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+ if (slot_to_key[slot] != N) { ok = false; break; }
+ for (std::size_t k2 = 0; k2 < k; ++k2) {
+ if (bucket_slots[k2] == slot) { ok = false; break; }
+ }
+ if (!ok) { break; }
+ bucket_slots[k] = slot;
+ }
+ if (ok) {
+ asso_values[0][ch] = d;
+ for (std::size_t k = 0; k < bk_count; ++k) {
+ slot_to_key[bucket_slots[k]] = bucket_keys[k];
+ }
+ placed = true;
+ break;
+ }
+ }
+ if (!placed) { return false; }
+ }
+ std::size_t filled = 0;
+ for (std::size_t i = 0; i < M; ++i) {
+ if (slot_to_key[i] != N) { ++filled; }
+ }
+ return filled == N;
+ };
+
+ if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+ positions[2] = HD_HASH_2BYTE_FLAG;
+ return true;
+ }
+ if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+ positions[2] = HD_HASH_4BYTE_FLAG;
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+ std::size_t npos{};
+ std::array<std::size_t, MAX_POSITIONS> pos{};
+ std::array<std::size_t, M> s2k{};
+ if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+ result.table_size = M;
+ result.asso_values = asso;
+ result.num_positions = npos;
+ result.positions = pos;
+ for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+ for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+ return true;
+ }
+ return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+ static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+ phf_result<N> result{};
+ if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+ constexpr std::size_t NextM = M * 2;
+ if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+ return compute_phf_hd_po2<N, NextM>(keys);
+ } else {
+ compile_time_error("Hash-and-Displace: failed to find valid table size");
+ return result;
+ }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+ constexpr std::size_t StartM = next_power_of_2(N);
+ constexpr std::size_t GPERF_MAX_TABLE =
+ phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+ if constexpr (StartM <= GPERF_MAX_TABLE) {
+ phf_result<N> result{};
+ if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+ }
+ return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+ std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+ std::array<std::uint8_t, MAX_POSITIONS> positions{};
+ std::uint8_t num_positions{};
+ std::uint8_t hd_hash_variant{}; // 2 or 4 (H&D only)
+ std::array<std::uint8_t, TableSize> slot_to_key{};
+ // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+ // multiple so the SIMD comparison can read a whole register.
+ std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+ std::array<std::uint8_t, TableSize> slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+ std::size_t m = 0;
+ for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+ return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+ for (std::size_t i = 0; i < N; ++i) {
+ if (keys[i].empty()) { compile_time_error("empty keys are not allowed in key_selector"); }
+ if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+ for (char c : keys[i]) {
+ if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+ if (c == '"') { compile_time_error("quote not allowed in key_selector keys"); }
+ if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+ }
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+ }
+ }
+
+ phf_data<N, TableSize, MaxKeyLen> out{};
+
+ if (result.num_positions == HD_MODE) {
+ // H&D mode: single displacement table in asso_values[0].
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+ }
+ out.num_positions = static_cast<std::uint8_t>(HD_MODE);
+ out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+ } else {
+ for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+ for (std::size_t c = 0; c < 256; ++c) {
+ out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+ }
+ }
+ out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+ for (std::size_t i = 0; i < result.num_positions; ++i) {
+ out.positions[i] = (result.positions[i] == LAST_CHAR)
+ ? POS_LAST_CHAR
+ : static_cast<std::uint8_t>(result.positions[i]);
+ }
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+ }
+
+ for (std::size_t s = 0; s < TableSize; ++s) {
+ std::size_t ki = result.slot_to_key[s];
+ if (ki < N) {
+ auto k = keys[ki];
+ out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+ for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+ } else {
+ out.slot_key_len[s] = 0; // empty slot: no length can match
+ }
+ }
+ return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+ "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+ return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+ // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+ // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+ // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+ // the SIMD and scalar builds agree.
+ static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+ uint64_t m = vget_lane_u64(
+ vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+ if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+ unsigned m = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i v = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+ // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+ // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+ unsigned m = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+ if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+ }
+ return MaxKeyLen + 1;
+#else
+ for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+ if (p[i] == '"') return i;
+ return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+ const char* p, const char* stored, std::size_t len) noexcept {
+ // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+ // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+ {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+ if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+ uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+ return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+ __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+ return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 32) {
+ [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+ {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+ uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+ uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+ uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16), lenv);
+ uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+ uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+ uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+ return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+ __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+ __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+ __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)), lenv);
+ __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+ __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+ return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+ __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+ __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+ __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+ __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0), lenv);
+ __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+ __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+ __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+ unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+ unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+ return (mlo & mhi) == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else if constexpr (MaxKeyLen <= 64) {
+ // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+ // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+ // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+ [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+ uint8x16_t base = vld1q_u8(idx16);
+ uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+ uint8x16_t acc = vdupq_n_u8(0);
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+ uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+ uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+ uint8x16_t mask = vcltq_u8(idxv, lenv);
+ acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+ }
+ return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+ __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+ __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+ int eq = 0xFFFF;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+ __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+ __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+ __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+ eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+ }
+ return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+ __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+ __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+ unsigned acc = 0xFFFFu;
+ for (std::size_t b = 0; b < num_blocks; ++b) {
+ __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+ __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+ __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+ __m128i mask = __lsx_vslt_b(idxv, lenv);
+ __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+ acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+ }
+ return acc == 0xFFFFu;
+#else
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+#endif
+ } else {
+ for (std::size_t i = 0; i < len; ++i)
+ if (p[i] != stored[i]) return false;
+ return true;
+ }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+ static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+ bool ok{false};
+ std::uint8_t byte_offset{0}; // first byte of the 2-byte read
+ std::uint8_t shift{0}; // intra-byte bit shift (0..7)
+ std::array<std::uint8_t, 256> window_to_key{}; // window byte -> key index, N if none
+ std::array<std::uint8_t, N> key_len{};
+ std::array<std::array<char, KEY_STRIDE>, N> key_bytes{}; // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t idx) noexcept {
+ if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+ return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+ std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+ unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+ unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+ return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+ window_data<N, MaxKeyLen> out{};
+
+ std::size_t min_len = keys[0].size();
+ for (std::size_t i = 1; i < N; ++i) {
+ if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+ }
+
+ // Iterate windows nearest the front first (cheapest to read, smallest shift).
+ for (std::size_t off = 0; off <= min_len; ++off) {
+ for (std::size_t shift = 0; shift < 8; ++shift) {
+ // The read touches byte off, and byte off+1 when shift != 0. Both must
+ // stay within the safe region [0, min_len] (min_len is the shortest
+ // key's quote index). off <= min_len is guaranteed by the loop bound.
+ if (shift != 0 && off + 1 > min_len) { continue; }
+
+ bool distinct = true;
+ for (std::size_t i = 0; i < N && distinct; ++i) {
+ for (std::size_t j = i + 1; j < N; ++j) {
+ if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+ distinct = false;
+ break;
+ }
+ }
+ }
+ if (!distinct) { continue; }
+
+ out.ok = true;
+ out.byte_offset = static_cast<std::uint8_t>(off);
+ out.shift = static_cast<std::uint8_t>(shift);
+ for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+ for (std::size_t i = 0; i < N; ++i) {
+ out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+ out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+ for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+ }
+ return out;
+ }
+ }
+ return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+ std::size_t shift) noexcept {
+ std::uint16_t w;
+ // Two controlled bytes (within the shortest key + its quote, hence within the
+ // padded buffer). memcpy is the portable little-endian unaligned load.
+ std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+ w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+ return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+ const window_data<N, MaxKeyLen>& w,
+ std::index_sequence<Is...>) noexcept {
+ std::size_t result = N;
+ auto try_match = [&](auto Ic) {
+ constexpr std::size_t i = decltype(Ic)::value;
+ if (ki == i && p[w.key_len[i]] == '"' &&
+ key_selector_detail::compare_key_bytes<MaxKeyLen>(
+ p, w.key_bytes[i].data(), w.key_len[i])) {
+ result = i;
+ }
+ };
+ (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+ return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+ if (v == 0) { s.push_back('0'); return; }
+ char buf[20];
+ std::size_t n = 0;
+ while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+ while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+ append_uint(s, b);
+ if (b >= 0x20 && b < 0x7f) {
+ s += " ('";
+ s.push_back(static_cast<char>(b));
+ s += "')";
+ }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ * using sel_t = key_selector<"id", "text", "user">;
+ * std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ * - Each key must be at most 63 characters long (and no longer than
+ * SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ * - The number of keys should be moderate. The hard limit is 255 keys;
+ * compilation time grows with the number of keys, so prefer a few dozen at
+ * most per selector.
+ * - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+ static constexpr std::size_t N = sizeof...(Keys);
+ static_assert(N > 0, "key_selector requires at least one key");
+ static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+ static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+ static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+ static_assert(max_key_len <= SIMDJSON_PADDING,
+ "key longer than SIMDJSON_PADDING is not supported");
+ // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+ // stays within the 64-byte padding guarantee. A 64-character key's closing
+ // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+ // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+ static_assert(max_key_len <= 63,
+ "key_selector keys must be at most 63 characters long");
+
+ static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+ static constexpr std::size_t table_size = result.table_size;
+
+ static constexpr auto phf =
+ key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+ // Single 8-bit-window discriminator (when one exists). Detected at compile
+ // time and selected with `if constexpr` below, so the hash path is compiled
+ // out for key sets that qualify, and this is compiled out for those that do
+ // not.
+ static constexpr auto window =
+ key_selector_detail::compute_window<N, max_key_len>(keys);
+
+ static constexpr std::size_t size() noexcept { return N; }
+
+ /**
+ * Look up a JSON key whose length is already known. p must point at the first
+ * key byte (just after the opening quote) in a padded simdjson buffer, and len
+ * must be the number of raw key bytes (the distance to the closing quote).
+ * Returns the selector index in [0, N) on match, or N on miss.
+ *
+ * Prefer this overload when the caller can obtain the key length cheaply (for
+ * example, object::for_each derives it from the structural index rather than
+ * re-scanning for the closing quote).
+ */
+ static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+ if (len == 0 || len > max_key_len) { return N; }
+
+ if constexpr (window.ok) {
+ // One 8-bit window selects the only possible candidate key;
+ // match_window_candidate confirms it (bytes + closing quote). p sits
+ // in a padded buffer and the window stays within the shortest key +
+ // quote, so the two-byte read is always in bounds. len is unused here
+ // because the quote check already pins the key's end.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+
+ std::size_t slot;
+ if (phf.num_positions == key_selector_detail::HD_MODE) {
+ // Hash-and-Displace: bucket displacement + per-key hash.
+ std::string_view key(p, len);
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(key)
+ : key_selector_detail::hd_key_hash_4(key);
+ slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ } else {
+ // gperf: h = len + sum of asso_values over the selected positions.
+ // positions / num_positions / asso_values are compile-time constants,
+ // so this loop fully unrolls. The idx < len guard mirrors the
+ // generator's char_at()-> 256 -> skip behavior for out-of-range
+ // positions (required: arbitrary positions may exceed a key's length).
+ std::size_t h = len;
+ for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+ std::uint8_t pos = phf.positions[i];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (len - std::size_t{1})
+ : static_cast<std::size_t>(pos);
+ if (idx < len) {
+ h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+ }
+ }
+ slot = h & (table_size - 1);
+ }
+
+ std::uint8_t ki = phf.slot_to_key[slot];
+ if (ki >= N) { return N; }
+ if (phf.slot_key_len[slot] != len) { return N; }
+ if (!key_selector_detail::compare_key_bytes<max_key_len>(
+ p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+ return ki;
+ }
+
+ /**
+ * Look up a JSON key. rjs must point just after an opening quote in a padded
+ * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+ * The key length is recovered with a SIMD scan for the closing quote; callers
+ * that already know the length should use the (p, len) overload above.
+ */
+ static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+ const char* p = rjs.raw();
+ if constexpr (window.ok) {
+ // One 8-bit window picks the candidate; verifying the candidate's
+ // bytes and its closing '"' confirms the full key, so the length scan
+ // is unnecessary. The window read is in bounds (padding), and the
+ // candidate length is at most max_key_len.
+ std::uint8_t ki = window.window_to_key[
+ key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+ if (ki >= N) { return N; }
+ return key_selector_detail::match_window_candidate(
+ p, ki, window, std::make_index_sequence<N>{});
+ }
+ return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+ }
+
+ /** Return the key text at selector index i (i in [0, N)). */
+ static constexpr std::string_view key_at(std::size_t i) noexcept {
+ return keys[i];
+ }
+
+ /**
+ * Return a complete, human-readable, multi-line description of how this
+ * selector classifies a key: which algorithm was selected at compile time
+ * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+ * exact bytes/positions it inspects, and the contents of the lookup tables
+ * (which window bytes or hash slots map to which key). The text mirrors what
+ * match_raw() does step by step.
+ *
+ * Everything it reports is derived from the compile-time tables, so describe()
+ * is itself usable in a constant expression when the standard library supports
+ * constexpr std::string (__cpp_lib_constexpr_string):
+ *
+ * static_assert(!key_selector<"name", "city">::describe().empty());
+ *
+ * It allocates a std::string and is meant for documentation, debugging and
+ * tests, not for any hot path.
+ */
+ static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+ std::string s;
+ s += "key_selector: ";
+ key_selector_detail::append_uint(s, N);
+ s += " keys, max key length ";
+ key_selector_detail::append_uint(s, max_key_len);
+ s += "\nkeys:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ s += " [";
+ key_selector_detail::append_uint(s, i);
+ s += "] \"";
+ s += keys[i];
+ s += "\" (length ";
+ key_selector_detail::append_uint(s, keys[i].size());
+ s += ")\n";
+ }
+ if constexpr (window.ok) {
+ // Mirrors the window fast path of match_raw().
+ s += "algorithm: single 8-bit window\n";
+ s += " step 1: read 2 bytes at offset ";
+ key_selector_detail::append_uint(s, window.byte_offset);
+ s += ", interpret them as a little-endian 16-bit value, shift right by ";
+ key_selector_detail::append_uint(s, window.shift);
+ s += " bits, and keep the low 8 bits\n";
+ s += " step 2: map that byte through a 256-entry table to a key index (";
+ key_selector_detail::append_uint(s, N);
+ s += " means no match):\n";
+ for (std::size_t b = 0; b < 256; ++b) {
+ if (window.window_to_key[b] < N) {
+ s += " byte ";
+ key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+ s += " -> key ";
+ key_selector_detail::append_uint(s, window.window_to_key[b]);
+ s += "\n";
+ }
+ }
+ s += " step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+ } else {
+ // Mirrors the perfect-hash path of match_raw().
+ if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+ s += "algorithm: hash-and-displace perfect hash\n";
+ s += " step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+ s += " step 2: keyhash = base-31 rolling hash of the length and the first ";
+ key_selector_detail::append_uint(s, phf.hd_hash_variant);
+ s += " bytes\n";
+ s += " step 3: slot = (displacement[bucket] + keyhash) mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += "\n step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+ std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+ ? key_selector_detail::hd_key_hash_2(k)
+ : key_selector_detail::hd_key_hash_4(k);
+ std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": bucket=";
+ key_selector_detail::append_uint(s, bucket);
+ s += " displacement=";
+ key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+ s += " keyhash=";
+ key_selector_detail::append_uint(s, kh);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ } else {
+ s += "algorithm: gperf-style perfect hash over ";
+ key_selector_detail::append_uint(s, phf.num_positions);
+ s += " character position(s)\n";
+ s += " step 1: h = key length\n";
+ s += " step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+ for (std::size_t i = 0; i < phf.num_positions; ++i) {
+ s += " position ";
+ if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+ s += "last character";
+ } else {
+ s += "byte index ";
+ key_selector_detail::append_uint(s, phf.positions[i]);
+ }
+ s += "\n";
+ }
+ s += " step 3: slot = h mod ";
+ key_selector_detail::append_uint(s, table_size);
+ s += " (a power of two, applied as a bitmask)\n";
+ s += " step 4: slot_to_key[slot] gives the candidate key index\n";
+ s += " per-key derivation:\n";
+ for (std::size_t i = 0; i < N; ++i) {
+ std::string_view k = keys[i];
+ std::size_t h = k.size();
+ for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+ std::size_t pos = phf.positions[pi];
+ std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+ ? (k.size() - 1) : pos;
+ if (idx < k.size()) {
+ h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+ }
+ }
+ std::size_t slot = h & (table_size - 1);
+ s += " \"";
+ s += k;
+ s += "\": h=";
+ key_selector_detail::append_uint(s, h);
+ s += " slot=";
+ key_selector_detail::append_uint(s, slot);
+ s += "\n";
+ }
+ }
+ s += " occupied slots (slot -> key):\n";
+ for (std::size_t slot = 0; slot < table_size; ++slot) {
+ if (phf.slot_to_key[slot] < N) {
+ s += " slot ";
+ key_selector_detail::append_uint(s, slot);
+ s += " -> key ";
+ key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+ s += " (\"";
+ s += keys[phf.slot_to_key[slot]];
+ s += "\", length ";
+ key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+ s += ")\n";
+ }
+ }
+ s += " confirm the candidate by checking the key length matches and comparing the key bytes\n";
+ }
+ return s;
+ }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for rvv_vls */
/* including simdjson/generic/ondemand/object.h for rvv_vls: #include "simdjson/generic/ondemand/object.h" */
/* begin file simdjson/generic/ondemand/object.h for rvv_vls */
#ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -178238,6 +228681,7 @@ public:
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
/* amalgamation skipped (editor-only): #include <vector> */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
@@ -178248,6 +228692,114 @@ namespace simdjson {
namespace rvv_vls {
namespace ondemand {
+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+ error_code error{SUCCESS};
+ std::size_t matched_count{0};
+ constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ * - an invocable taking a value (run custom logic for that field), or
+ * - a deserialization target T, in which case the matched value is assigned
+ * directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+ std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+ ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+ std::is_invocable_v<std::remove_reference_t<H>&, value>
+ ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+ : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+ (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+ /**
+ * Default-constructed so a variable can be declared and assigned later,
+ * matching e.g. document()/object(). Not a valid position to revert to.
+ */
+ simdjson_inline object_position() noexcept = default;
+
+private:
+ token_position position{};
+ depth_t depth{};
+
+ simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+ : position(position_), depth(depth_) {}
+
+ friend class object;
+};
+
/**
* A forward-only JSON object field iterator.
*/
@@ -178266,8 +228818,19 @@ public:
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
* you must dereference the iterator exactly once per iteration (before calling '++').
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+ *
+ * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+ * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+ * OUT_OF_ORDER_ITERATION.
+ */
+ simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+ /**
+ * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+ *
+ * The iterator does not depend on the object instance and may outlive it, so
+ * it does not lock it.
*/
- simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<object_iterator> end() noexcept;
/**
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -178279,10 +228842,11 @@ public:
*
* ```cpp
* simdjson::ondemand::parser parser;
- * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
- * double z = obj.find_field("z");
- * double y = obj.find_field("y");
- * double x = obj.find_field("x");
+ * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+ * auto doc = parser.iterate(json);
+ * double z = doc.find_field("z");
+ * double y = doc.find_field("y");
+ * double x = doc.find_field("x");
* ```
* If you have multiple fields with a matching key ({"x": 1, "x": 1}) be mindful
* that only one field is returned.
@@ -178355,6 +228919,100 @@ public:
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ /**
+ * Walk this object once and invoke on_match(selector_index, value) for each
+ * field whose key is in the compile-time key_selector Selector, in JSON order
+ * (first occurrence of a duplicate key wins). Iteration stops once all
+ * Selector::size() keys have matched or the object ends. The value is consumed
+ * in place, so this is a low-overhead way to extract a known set of fields
+ * regardless of their order in the JSON.
+ *
+ * Like other object iteration in simdjson, for_each consumes the object by
+ * advancing the underlying iterator state; after the call the same object
+ * instance should not be used for further field access or iteration.
+ *
+ * Usage:
+ * using sel_t = ondemand::key_selector<"id", "text", "user">;
+ * obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+ * switch (i) { case 0: ...; case 1: ...; }
+ * });
+ *
+ * Limitations (see key_selector): each key must be at most 63 characters long,
+ * and the number of keys should be moderate (hard limit 255; a handful is
+ * best, as the compile-time perfect hash may fail or slow compilation for
+ * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+ * null bytes.
+ *
+ * The callback may return either void or an error_code. When it returns an
+ * error_code, the walk stops at the first non-SUCCESS result and that error is
+ * returned, which lets the callback surface value-parse errors.
+ *
+ * This function is conditionally noexcept: it is noexcept exactly when invoking
+ * the callback is noexcept. The callback runs inside this frame, so a throwing
+ * callback (e.g. one using the exception-throwing conversions like
+ * std::string_view(value) or uint64_t(value)) makes for_each potentially
+ * throwing too -- the exception propagates to the caller instead of crossing a
+ * noexcept boundary and calling std::terminate.
+ *
+ * @returns a for_each_result holding the first error encountered while walking
+ * the object (including any error returned by the callback, SUCCESS if
+ * none) and the number of distinct selector keys that matched. The
+ * result converts implicitly to error_code, so callers that only need
+ * the error can ignore the count.
+ */
+ template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+ simdjson_inline for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+ /**
+ * Variadic per-key form. Provide exactly one handler per key in the Selector
+ * (compiler-enforced). Handlers are processed in JSON document order for the
+ * matching keys. Each handler is either:
+ * - a deserialization target (a variable), in which case the matched value
+ * is assigned to it via value::get -- no lambda required; or
+ * - an invocable taking the ondemand::value (for custom logic such as
+ * descending into a nested object). It may return void or error_code;
+ * returning error_code lets you surface parse/type errors.
+ * The two styles may be mixed freely, one handler per key.
+ *
+ * Example (bind fields straight to variables):
+ * using fields = ondemand::key_selector<"name", "city", "age">;
+ * obj.for_each<fields>(name, city, age);
+ *
+ * Example (mixing a target and a lambda):
+ * obj.for_each<ondemand::key_selector<"id", "user">>(
+ * id, // assigned via value::get
+ * [&](ondemand::value v){ u = read_user(v); } // custom logic
+ * );
+ *
+ * The index-based single-callback form (taking (size_t, value)) remains
+ * available for shared-state or more complex per-key logic.
+ */
+ template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+ * Lets you write the keys inline without a separate using/alias, binding each
+ * field straight to a variable (or a lambda, see the Selector form above):
+ *
+ * obj.for_each<"name", "city", "age">(name, city, age);
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline for_each_result for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
/**
* Get the value associated with the given JSON pointer. We use the RFC 6901
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -178431,6 +229089,34 @@ public:
* @returns true if the object contains some elements (not empty)
*/
inline simdjson_result<bool> reset() & noexcept;
+ /**
+ * Get an opaque token representing the object's current scanning position.
+ * Pass it to revert_position() to return to this exact point later, without
+ * paying the cost of a full reset() and re-scan from the beginning.
+ *
+ * A typical use is an optional field that may or may not be next: capture
+ * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+ * instead of reset() so that fields already consumed are not rescanned.
+ *
+ * The returned token is only valid for this object, and only until it is
+ * reset() or the parser iterate()s a new document; using it after either
+ * is undefined behavior (see object_position).
+ *
+ * @returns An opaque position token.
+ */
+ simdjson_inline object_position get_current_position() const noexcept;
+ /**
+ * Return the object's scanning position to a snapshot previously obtained
+ * from get_current_position(). Unlike reset(), this does not rescan the
+ * object from the beginning: fields before the captured position remain
+ * consumed, and scanning resumes exactly where the snapshot was captured.
+ *
+ * @param position A snapshot previously returned by get_current_position(),
+ * for this same object.
+ * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+ * iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+ */
+ simdjson_inline error_code revert_position(object_position position) noexcept;
/**
* This method scans the beginning of the object and checks whether the
* object is empty.
@@ -178476,7 +229162,7 @@ public:
*/
template <typename T>
simdjson_warn_unused simdjson_inline error_code get(T &out)
- noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+ noexcept(nothrow_gettable<T, object>) {
static_assert(custom_deserializable<T, object>);
return deserialize(*this, out);
}
@@ -178488,7 +229174,7 @@ public:
*/
template <typename T>
simdjson_inline simdjson_result<T> get()
- noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+ noexcept(nothrow_gettable<T, object>)
{
static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
T out{};
@@ -178540,10 +229226,18 @@ protected:
simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;
value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ bool locked{false};
+ simdjson_inline void set_locked(bool _locked) noexcept;
+#endif
friend class value;
friend class document;
friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ friend class object_iterator;
+ friend struct simdjson_result<object_iterator>;
+#endif
};
} // namespace ondemand
@@ -178559,7 +229253,8 @@ public:
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
simdjson_inline simdjson_result() noexcept = default;
- simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() noexcept;
+ simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() & noexcept;
+ simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() && noexcept;
simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> end() noexcept;
simdjson_inline simdjson_result<rvv_vls::ondemand::value> find_field(std::string_view key) & noexcept;
simdjson_inline simdjson_result<rvv_vls::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -178577,6 +229272,8 @@ public:
#endif
simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
inline simdjson_result<bool> reset() noexcept;
+ inline simdjson_result<rvv_vls::ondemand::object_position> get_current_position() noexcept;
+ inline error_code revert_position(rvv_vls::ondemand::object_position position) noexcept;
inline simdjson_result<bool> is_empty() noexcept;
inline simdjson_result<size_t> count_fields() & noexcept;
inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -178584,7 +229281,7 @@ public:
// TODO: move this code into object-inl.h
template<typename T>
- simdjson_inline simdjson_result<T> get() noexcept {
+ simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, rvv_vls::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, rvv_vls::ondemand::object>) {
return first;
@@ -178592,7 +229289,7 @@ public:
return first.get<T>();
}
template<typename T>
- simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+ simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, rvv_vls::ondemand::object>) {
if (error()) { return error(); }
if constexpr (std::is_same_v<T, rvv_vls::ondemand::object>) {
out = first;
@@ -178602,6 +229299,39 @@ public:
return SUCCESS;
}
+ /**
+ * Forwards to object::for_each on the underlying object, so error-code-style
+ * chains (e.g. doc["x"].get_object()) can call for_each without first
+ * extracting the object. If this result holds an error, that error is returned
+ * (with a zero match count) and the callback is not invoked. See
+ * object::for_each for the semantics.
+ */
+ template <typename Selector, typename Func>
+ requires rvv_vls::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>
+ simdjson_inline rvv_vls::ondemand::for_each_result for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>);
+
+ /**
+ * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+ */
+ template <typename Selector, typename... Handlers>
+ requires rvv_vls::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline rvv_vls::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+ /**
+ * Forwarding overload for the direct-key variadic form.
+ */
+ template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+ simdjson_inline rvv_vls::ondemand::for_each_result for_each(Handlers&&... on_match)
+ noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
#if SIMDJSON_STATIC_REFLECTION
// TODO: move this code into object-inl.h
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -178642,6 +229372,15 @@ public:
*/
simdjson_inline object_iterator() noexcept = default;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ simdjson_inline ~object_iterator() noexcept;
+
+ simdjson_inline object_iterator(object_iterator&&) noexcept;
+ simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+ simdjson_inline object_iterator(const object_iterator&) noexcept;
+ simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
//
// Iterator interface
//
@@ -178661,6 +229400,9 @@ public:
private:
#if SIMDJSON_DEVELOPMENT_CHECKS
bool has_been_referenced{false};
+ object* parent{nullptr};
+
+ simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
#endif
/**
* The underlying JSON iterator.
@@ -178706,6 +229448,191 @@ public:
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
/* end file simdjson/generic/ondemand/object_iterator.h for rvv_vls */
+/* including simdjson/generic/ondemand/ranges.h for rvv_vls: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<value>;
+ using reference = simdjson_result<value>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline array_range_iterator() noexcept = default;
+ simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+ /**
+ * Get the current element. Const-qualified for std::indirectly_readable;
+ * internally delegates to the mutable wrapped iterator.
+ */
+ simdjson_inline simdjson_result<value> operator*() const noexcept;
+ simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ /**
+ * Comparison delegates to array_iterator::operator==, which checks
+ * whether the underlying parser has finished the array (depth-based).
+ */
+ simdjson_inline friend bool operator==(const array_range_iterator& a,
+ const array_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ * ondemand::parser parser;
+ * auto doc = parser.iterate(json);
+ * auto arr = doc.get_array().value();
+ * for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+ simdjson_inline array_range() noexcept = default;
+ simdjson_inline explicit array_range(array& arr) noexcept;
+
+ simdjson_inline array_range_iterator begin() noexcept;
+ simdjson_inline array_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ array_iterator begin_{};
+ array_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+ using iterator_concept = std::input_iterator_tag;
+ using value_type = simdjson_result<field>;
+ using reference = simdjson_result<field>;
+ using difference_type = std::ptrdiff_t;
+
+ simdjson_inline object_range_iterator() noexcept = default;
+ simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+ simdjson_inline simdjson_result<field> operator*() const noexcept;
+ simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+ simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+ simdjson_inline friend bool operator==(const object_range_iterator& a,
+ const object_range_iterator& b) noexcept {
+ return a.iter_ == b.iter_;
+ }
+
+private:
+ mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+ simdjson_inline object_range() noexcept = default;
+ simdjson_inline explicit object_range(object& obj) noexcept;
+
+ simdjson_inline object_range_iterator begin() noexcept;
+ simdjson_inline object_range_iterator end() noexcept;
+
+ /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+ simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+ object_iterator begin_{};
+ object_iterator end_{};
+ error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::rvv_vls::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::rvv_vls::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for rvv_vls */
/* including simdjson/generic/ondemand/serialization.h for rvv_vls: #include "simdjson/generic/ondemand/serialization.h" */
/* begin file simdjson/generic/ondemand/serialization.h for rvv_vls */
#ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -178838,12 +229765,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <concepts>
#include <limits>
#if SIMDJSON_STATIC_REFLECTION
#include <meta>
+#include <vector>
// #include <static_reflection> // for std::define_static_string - header not available yet
#endif
@@ -178868,10 +229797,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
template <std::floating_point T>
error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
- double x;
- SIMDJSON_TRY(val.get_double().get(x));
- out = static_cast<T>(x);
- return SUCCESS;
+ if constexpr (std::is_same_v<T, float>) {
+ // Going through binary64 and then rounding to binary32 would round twice
+ // and could produce a value that is not the float nearest to the JSON
+ // number, so we parse to binary32 directly.
+ return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+ } else if constexpr (std::is_same_v<T, std::float32_t>) {
+ // Same reason as float.
+ float x;
+ SIMDJSON_TRY(val.get_float().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+ } else {
+ double x;
+ SIMDJSON_TRY(val.get_double().get(x));
+ out = static_cast<T>(x);
+ return SUCCESS;
+ }
}
template <std::signed_integral T>
@@ -178907,11 +229851,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- out = T{str};
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::string): building a temporary and
+ // move-assigning it is markedly slower.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
+ return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+ std::u8string_view str;
+ SIMDJSON_TRY(val.get_u8string().get(str));
+ if constexpr (requires { out.assign(str.data(), str.size()); }) {
+ // Copy straight into out (e.g., std::u8string), as for std::string above.
+ out.assign(str.data(), str.size());
+ } else {
+ out = T{str};
+ }
return SUCCESS;
}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+ concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+ !std::is_trivially_copyable_v<typename T::value_type> &&
+ (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+ T &container;
+ bool armed{true};
+ ~pop_back_guard() {
+ if (armed) { container.pop_back(); }
+ }
+};
+} // namespace details
+
/**
* STL containers have several constructors including one that takes a single
* size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -178935,22 +229927,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
SIMDJSON_TRY(val.get_array().get(arr));
}
- for (auto v : arr) {
- if constexpr (concepts::returns_reference<T>) {
- if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
- err) {
- // If an error occurs, the empty element that we just inserted gets
- // removed. We're not using a temp variable because if T is a heavy
- // type, we want the valid path to be the fast path and the slow path be
- // the path that has errors in it.
- if constexpr (requires { out.pop_back(); }) {
- static_cast<void>(out.pop_back());
+ if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+ // Collect the elements in a per-thread scratch vector that keeps its
+ // capacity from call to call, then move them into out after reserving the
+ // exact size: out is allocated once instead of being regrown. A nested
+ // array of the same type finds the scratch busy and takes the paths below.
+ // Prior related work: jsonifier keeps a thread-local vector and sizes the
+ // caller's vector from that element count (parse_impl.hpp,
+ // https://github.com/nihilai-collective/Jsonifier).
+ struct scratch_space {
+ std::vector<value_type> elements{};
+ bool busy{false};
+ };
+ static thread_local scratch_space scratch;
+ if (!scratch.busy && out.empty()) {
+ struct release_scratch {
+ scratch_space &s;
+ T &out;
+ size_t parsed{0};
+ bool complete{false};
+ // On an error or an exception, out gets the elements parsed so far (as
+ // with the loops below), without allocating. Kept out of the hot path.
+ simdjson_never_inline void keep_parsed() noexcept {
+ s.elements.resize(parsed);
+ out.swap(s.elements);
}
- return err;
- }
- } else {
+ ~release_scratch() {
+ if (simdjson_unlikely(!complete)) { keep_parsed(); }
+ s.elements.clear();
+ // Do not hold on to the memory of a very large array.
+ if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+ s.busy = false;
+ }
+ } release{scratch, out};
+ scratch.busy = true;
+ for (auto v : arr) {
+ SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+ release.parsed++;
+ }
+ out.reserve(release.parsed);
+ release.complete = true;
+ for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+ return SUCCESS;
+ }
+ }
+ if constexpr (details::deserialize_in_place<T>) {
+ for (auto v : arr) {
+ auto &slot = concepts::emplace_one(out);
+ // An error or an exception (a user tag_invoke may throw) must not leave
+ // a partially deserialized element behind.
+ details::pop_back_guard<T> guard{out};
+ SIMDJSON_TRY(v.get<value_type>(slot));
+ guard.armed = false;
+ }
+ } else {
+ for (auto v : arr) {
+ // Deserialize into a temporary first: an error or an exception (a user
+ // tag_invoke may throw) must not leave a default-constructed element behind.
value_type temp;
- if (auto const err = v.get<value_type>().get(temp); err) {
+ if (auto const err = v.get<value_type>(temp); err) {
return err;
}
concepts::emplace_one(out, std::move(temp));
@@ -178991,7 +230026,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
using value_type = typename std::remove_cvref_t<T>::mapped_type;
out.clear();
@@ -179010,21 +230045,21 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) n
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::value &val, T &out) noexcept(false) {
rvv_vls::ondemand::object obj;
SIMDJSON_TRY(val.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document &doc, T &out) noexcept(false) {
rvv_vls::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
}
template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &doc, T &out) noexcept(false) {
rvv_vls::ondemand::object obj;
SIMDJSON_TRY(doc.get_object().get(obj));
return simdjson::deserialize(obj, out);
@@ -179035,10 +230070,6 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &do
* This CPO (Customization Point Object) will help deserialize into
* smart pointers.
*
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
* @tparam T The type inside the smart pointer
* @tparam ValT document/value type
* @param val document/value
@@ -179046,7 +230077,7 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &do
* @return status of the conversion
*/
template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
using element_type = typename std::remove_cvref_t<T>::element_type;
// For better error messages, don't use these as constraints on
@@ -179058,12 +230089,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
std::is_default_constructible_v<element_type>,
"The specified type inside the unique_ptr must default constructible.");
- auto ptr = new (std::nothrow) element_type();
- if (ptr == nullptr) {
+ // Own the allocation before get(): a user tag_invoke may throw.
+ std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+ if (!ptr) {
return MEMALLOC;
}
SIMDJSON_TRY(val.template get<element_type>(*ptr));
- out.reset(ptr);
+ out = std::move(ptr);
return SUCCESS;
}
@@ -179095,53 +230127,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser
template <typename T>
constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, rvv_vls::ondemand::number>
+&& !std::is_same_v<T, rvv_vls::ondemand::document>
+&& !std::is_same_v<T, rvv_vls::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+ return !std::meta::is_const(mem) && std::meta::is_public(mem)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+ // The data member holding the value; its annotations drive (de)serialization.
+ static constexpr std::meta::info leaf = [] {
+ std::meta::info members[] = {First, Rest...};
+ return members[sizeof...(Rest)];
+ }();
+ template <typename T>
+ static simdjson_inline constexpr auto &get(T &obj) noexcept {
+ if constexpr (sizeof...(Rest) == 0) {
+ return obj.[:First:];
+ } else {
+ return member_path<Rest...>::get(obj.[:First:]);
+ }
+ }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+ && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+ std::vector<std::meta::info> &fields) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (!is_eligible_member(mem)) { continue; }
+ prefix.push_back(std::meta::reflect_constant(mem));
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ std::meta::info flattened = simdjson::detail::flattened_type(mem);
+ if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+ throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+ }
+ append_eligible_fields(flattened, prefix, fields);
+ } else {
+ fields.push_back(std::meta::substitute(^^member_path, prefix));
+ }
+ prefix.pop_back();
+ }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+ std::vector<std::meta::info> prefix;
+ std::vector<std::meta::info> fields;
+ append_eligible_fields(type, prefix, fields);
+ return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+ return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+ return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+ std::vector<const char *> keys;
+ for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+ bool repeated = false;
+ for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+ if (!repeated) { keys.push_back(std::define_static_string(key)); }
+ }
+ return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ for (std::meta::info path : eligible_fields(type)) {
+ for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+ }
+ return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+ std::vector<std::size_t> key_fields;
+ std::vector<std::meta::info> fields = eligible_fields(type);
+ for (std::size_t i = 0; i < fields.size(); ++i) {
+ for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+ }
+ return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+ std::vector<const char *> keys = accepted_keys(^^T);
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+ for (std::string_view key : accepted_keys(^^T)) {
+ for (char c : key) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+ }
+ }
+ return false;
+}
+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+ return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+ return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+ || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+ return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+ bool all_required = true;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ if constexpr (may_be_absent<[: path :]::leaf>()) {
+ all_required = false;
+ }
+ }
+ return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+ constexpr std::string_view key_view{ key };
+ char buffer[key_view.size() + 1] = {};
+ for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+ return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+ std::vector<std::meta::info> args;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+ args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+ }
+ return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+ ^^rvv_vls::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+ std::vector<std::string_view> keys;
+ for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+ if (keys.size() > 255) { return false; }
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (keys[i].empty() || keys[i].size() > 63) { return false; }
+ for (char c : keys[i]) {
+ if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+ }
+ for (std::size_t j = i + 1; j < keys.size(); ++j) {
+ if (keys[i] == keys[j]) { return false; }
+ }
+ }
+ return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+ for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+ if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+ }
+ return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+ constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+ if constexpr (with_type != std::meta::info{}) {
+ using adapter = typename [: with_type :]::adapter;
+ using ondemand_value = rvv_vls::ondemand::value;
+ if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+ return adapter::deserialize(field_value, target);
+ } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+ && requires { field_value.get_value(); }) {
+ // A transparent structure read from a document: the adapter takes an
+ // ondemand::value. A scalar document cannot be viewed as a value, so it
+ // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+ // document itself and has no such limitation).
+ ondemand_value v;
+ SIMDJSON_TRY(field_value.get_value().get(v));
+ return adapter::deserialize(v, target);
+ } else {
+ static_assert(!declares_deserialize(^^adapter),
+ "the deserialize function of a simdjson::with adapter must be callable as "
+ "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+ return field_value.get(target);
+ }
+ } else {
+ return field_value.get(target);
+ }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+ if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+ // A present key replaces the default value: deserialize into a fresh
+ // temporary so that, e.g., a container does not append to its default
+ // content, and a failure leaves the default untouched.
+ M value{};
+ SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+ target = std::move(value);
+ return SUCCESS;
+ } else {
+ return deserialize_member_value<mem>(field_value, target);
+ }
+}
+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+ std::size_t field_index, rvv_vls::ondemand::value field_value, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+ constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+ if constexpr (default_from_type != std::meta::info{}) {
+ target = [: default_from_type :]::factory();
+ return SUCCESS;
+ } else if constexpr (may_be_absent<mem>()) {
+ // For optional and default_value members, a missing key is not an error:
+ // leave the member at its current (default) value.
+ (void)target;
+ return SUCCESS;
+ } else {
+ (void)target;
+ return NO_SUCH_FIELD;
+ }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+ const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+ ++counter;
+ }
+ return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+ rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
+ template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+ using field = [: path :];
+ rvv_vls::ondemand::value field_value;
+ error_code error = NO_SUCH_FIELD;
+ template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+ if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+ }
+ if (error == NO_SUCH_FIELD) {
+ SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+ } else if (error) {
+ return error;
+ } else {
+ SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+ }
+ }
+ return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+ for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+ || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+ continue;
+ }
+ bool unassignable = all || !is_eligible_member(mem);
+ if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+ append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+ } else if (unassignable) {
+ keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+ }
+ }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+ std::vector<const char *> keys;
+ append_unassignable_keys(type, false, keys);
+ return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+ rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
+ static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+ static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+ std::array<bool, eligible_field_count<T>()> seen_field{};
+ for (auto field_result : obj) {
+ rvv_vls::ondemand::field json_field;
+ SIMDJSON_TRY(std::move(field_result).get(json_field));
+ std::string_view key;
+ SIMDJSON_TRY(json_field.unescaped_key().get(key));
+ std::size_t key_index = keys.size();
+ for (std::size_t i = 0; i < keys.size(); ++i) {
+ if (key == std::string_view(keys[i])) { key_index = i; break; }
+ }
+ if (key_index == keys.size()) {
+ if constexpr (DenyUnknown) {
+ // A key that T itself serializes (e.g. of a const member) is not
+ // unknown: a serialized value must parse back.
+ static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+ bool ignored = false;
+ for (const char *ignored_key : ignored_keys) {
+ if (key == std::string_view(ignored_key)) { ignored = true; break; }
+ }
+ if (!ignored) { return UNKNOWN_FIELD; }
+ }
+ continue;
+ }
+ const std::size_t field_index = key_fields[key_index];
+ if (seen_field[field_index]) { continue; }
+ seen_field[field_index] = true;
+ SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+ }
+ return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+ return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+// - globally, the ordered per-member path when defining
+// -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+// - automatically and per-type, a scan of the object comparing unescaped keys
+// when the struct's keys do not fit the key_selector limits (see
+// keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+// so that long member names and the like keep compiling rather than
+// tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
template <typename T, typename ValT>
requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+ if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+ constexpr auto mem = simdjson::detail::transparent_member(^^T);
+ if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, rvv_vls::ondemand::object>) {
+ // We were handed an object: only a structure can be deserialized from it.
+ if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+ return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+ } else {
+ return INCORRECT_TYPE;
+ }
+ } else {
+ return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+ }
+ } else {
+ static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+ "two members of this structure accept the same JSON key (check rename, alias, "
+ "rename_all and flatten)");
rvv_vls::ondemand::object obj;
if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, rvv_vls::ondemand::object>) {
obj = val;
} else {
SIMDJSON_TRY(val.get_object().get(obj));
}
- template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
- if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
- constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
- if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
- // for optional members, it's ok if the key is missing
- auto error = obj[key].get(out.[:mem:]);
- if (error && error != NO_SUCH_FIELD) {
- if(error == NO_SUCH_FIELD) {
- out.[:mem:].reset();
- continue;
- }
- return error;
- }
- } else {
- // for non-optional members, the key must be present
- SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+ if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+ return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+ } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+ // match T's keys.
+ if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+ }
+#else
+ if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+ // No fields to deserialize: an empty key_selector cannot be built, so just
+ // validate that the input is an object (done above) and succeed. Mirrors the
+ // ordered per-member path, which iterates over zero members.
+ (void)out;
+ (void)obj;
+ return SUCCESS;
+ } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+ // Automatic fallback: T's accepted keys do not fit the key_selector limits
+ // (e.g. a member name longer than 63 characters, or a key with a double
+ // quote), so building a selector would be a compile error. Scan the object
+ // instead, so the default never breaks a struct that the opt-out path would
+ // accept.
+ return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+ } else {
+ using selector = key_selector_reflection_detail::selector_for<T>;
+ if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+ && !key_selector_reflection_detail::has_aliases<T>()) {
+ // Fast path: every member is required and has a single key. A single
+ // for_each pass parses each matched field; the returned match count then
+ // tells us whether every member was present (matched_count ==
+ // selector::size()) without a per-member "seen" array. A value-parse error
+ // (e.g. a type mismatch) is propagated by for_each.
+ auto walk = obj.template for_each<selector>(
+ [&](std::size_t matched_index, rvv_vls::ondemand::value field_value) -> error_code {
+ std::size_t counter = 0;
+ template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+ using field = [: path :];
+ if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+ ++counter;
}
- }
- };
- return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+ return SUCCESS;
+ });
+ if (walk.error) { return walk.error; }
+ // A missing required member shows up as a short match count and is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+ if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+ return SUCCESS;
+ } else {
+ static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+ std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+ // Single pass over the object: each field whose key matches a member (or one
+ // of its aliases) yields its selector index, which we map back to the
+ // corresponding member. The first key seen for a member wins. The callback
+ // returns an error_code so that a value-parse error (e.g. a type mismatch on
+ // a matched field) is propagated by for_each instead of being silently dropped.
+ error_code walk_error = obj.template for_each<selector>(
+ [&](std::size_t matched_index, rvv_vls::ondemand::value field_value) -> error_code {
+ const std::size_t field_index = key_fields[matched_index];
+ if (seen_field[field_index]) { return SUCCESS; }
+ seen_field[field_index] = true;
+ return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+ });
+ if (walk_error) { return walk_error; }
+ // Required members must be present: a missing one is reported as
+ // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+ // members may be absent.
+ return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+ }
+ }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+ }
+ }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
template <typename T, typename ValT>
requires(std::is_enum_v<T>)
error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
#if SIMDJSON_STATIC_REFLECTION
std::string_view str;
SIMDJSON_TRY(val.get_string().get(str));
- constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+ static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
template for (constexpr auto enum_val : enumerators) {
- if (str == std::meta::identifier_of(enum_val)) {
- out = [:enum_val:];
- return SUCCESS;
+ template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+ if (str == std::string_view(key)) {
+ out = [:enum_val:];
+ return SUCCESS;
+ }
}
};
@@ -179157,33 +230731,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_unique<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+ std::unique_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
template <typename simdjson_value, typename T>
requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
- if (!out) {
- out = std::make_shared<T>();
- if (!out) {
- return MEMALLOC;
- }
- }
- if (auto err = val.get(*out)) {
- out.reset();
- return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+ std::shared_ptr<T> ptr(new (std::nothrow) T());
+ if (!ptr) {
+ return MEMALLOC;
}
+ SIMDJSON_TRY(val.get(*ptr));
+ out = std::move(ptr);
return SUCCESS;
}
@@ -179495,9 +231061,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
return array(iter);
}
-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return array_iterator(iter, this);
+#endif
+ return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The array is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return array_iterator(iter);
}
@@ -179524,6 +231098,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
SIMDJSON_PUSH_DISABLE_WARNINGS
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t count{0};
// Important: we do not consume any of the values.
for(simdjson_unused auto v : *this) { count++; }
@@ -179537,6 +231114,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
SIMDJSON_POP_DISABLE_WARNINGS
simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_array().get(is_not_empty);
if(error) { return error; }
@@ -179544,31 +231124,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
}
inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_array();
}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
// - means "the append position" or "the element after the end of the array"
// We don't support this, because we're returning a real element, not a position.
if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }
- // Read the array index
size_t array_index = 0;
size_t i;
- for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
- uint8_t digit = uint8_t(json_pointer[i] - '0');
- // Check for non-digit in array index. If it's there, we're trying to get a field in an object
- if (digit > 9) { return INCORRECT_TYPE; }
- array_index = array_index*10 + digit;
- }
-
- // 0 followed by other digits is invalid
- if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
- // Empty string is invalid; so is a "/" with no digits before it
- if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+ SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
// Get the child
auto child = at(array_index);
// If there is an error, it ends here
@@ -179642,6 +231221,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
}
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
size_t i = 0;
for (auto value : *this) {
if (i == index) { return value; }
@@ -179671,10 +231253,14 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::array>::simdjson_result(
{
}
-simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -179737,6 +231323,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
SIMDJSON_ASSUME(!has_been_referenced);
@@ -179832,6 +231471,41 @@ namespace simdjson {
namespace rvv_vls {
namespace ondemand {
+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+ uint64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+ int64_t result;
+ SIMDJSON_TRY(std::move(wide).get(result));
+ if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+ return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+ "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+ "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
simdjson_inline value::value(const value_iterator &_iter) noexcept
: iter{_iter}
{
@@ -179863,6 +231537,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
return iter.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
return iter.get_string(receiver, allow_replacement);
@@ -179876,6 +231557,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
return iter.get_double_in_string();
}
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+ return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+ return iter.get_float_in_string();
+}
simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
return iter.get_uint64();
}
@@ -179889,17 +231576,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
return iter.get_int64_in_string();
}
simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
return iter.get_bool();
}
@@ -179911,12 +231618,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }
@@ -179924,12 +231645,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }
#if SIMDJSON_EXCEPTIONS
@@ -180098,6 +231833,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
}
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+ // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+ // document::at_pointer.
+ if (json_pointer.empty()) { return value(iter); }
json_type t;
SIMDJSON_TRY(type().get(t));
switch (t)
@@ -180135,6 +231873,10 @@ template <typename Func>
template <typename Func>
#endif
inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+ // Every recursive step of for_each_at_path_with_wildcard goes through this
+ // function, and each one descends one level into the document. A path with
+ // many segments applied to a deeply nested document would otherwise recurse
+ // without bound and overflow the stack.
if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
json_type t;
SIMDJSON_TRY(type().get(t));
@@ -180248,10 +231990,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::valu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::value>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::value>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::value>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::value>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::value>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::value>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::value>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::value>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::value>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::value>::get_double_in_string() noexcept {
if (error()) { return error(); }
return first.get_double_in_string();
@@ -180260,6 +232038,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondem
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -180288,11 +232072,23 @@ template<> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>:
return SUCCESS;
}
-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
-template<typename T> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
@@ -180562,16 +232358,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
return get_root_value_iterator().get_root_int64_in_string(true);
}
simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(get_uint64().get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
+ return narrow_integer<uint32_t>(get_uint64());
}
simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(get_int64().get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
+ return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+ return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+ return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+ return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+ return narrow_integer<int8_t>(get_int64());
}
simdjson_inline simdjson_result<double> document::get_double() noexcept {
return get_root_value_iterator().get_root_double(true);
@@ -180579,9 +232381,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
return get_root_value_iterator().get_root_double_in_string(true);
}
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+ return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+ return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(true, allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -180603,11 +232432,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
@@ -180615,17 +232458,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -180964,6 +232825,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::document>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::document>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::document>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::document>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::document>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -180972,10 +232849,36 @@ simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::document>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::document>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::document>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -181003,22 +232906,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::documen
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<rvv_vls::ondemand::document>(first).get<T>();
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<rvv_vls::ondemand::document>(first).get<T>(out);
}
@@ -181087,27 +233014,27 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -181197,21 +233124,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
- uint64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
- if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
- int64_t result;
- SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
- if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
- return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+ float result;
+ SIMDJSON_TRY(get_float().get(result));
+ return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+ double result;
+ SIMDJSON_TRY(get_double().get(result));
+ return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+ std::string_view content;
+ SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+ return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -181223,11 +233167,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
#if SIMDJSON_EXCEPTIONS
@@ -181373,6 +233331,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::docu
if (error()) { return error(); }
return first.get_int32();
}
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_uint16() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_int16() noexcept {
+ if (error()) { return error(); }
+ return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_uint8() noexcept {
+ if (error()) { return error(); }
+ return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_int8() noexcept {
+ if (error()) { return error(); }
+ return first.get_int8();
+}
simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::document_reference>::get_double() noexcept {
if (error()) { return error(); }
return first.get_double();
@@ -181381,10 +233355,36 @@ simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::docum
if (error()) { return error(); }
return first.get_double_in_string();
}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document_reference>::get_float() noexcept {
+ if (error()) { return error(); }
+ return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document_reference>::get_float_in_string() noexcept {
+ if (error()) { return error(); }
+ return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_float32() noexcept {
+ if (error()) { return error(); }
+ return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_float64() noexcept {
+ if (error()) { return error(); }
+ return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.get_string(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
template <typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -181411,22 +233411,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::documen
return first.is_null();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>();
}
template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<rvv_vls::ondemand::document_reference>(first).get<T>();
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return first.get<T>(out);
}
template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+ noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+ noexcept
+#endif
+{
if (error()) { return error(); }
return std::forward<rvv_vls::ondemand::document_reference>(first).get<T>(out);
}
@@ -181488,27 +233512,27 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator uint64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_uint64();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator int64_t() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_int64();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator double() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_double();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator std::string_view() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_string();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator rvv_vls::ondemand::raw_json_string() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_raw_json_string();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator bool() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
- return first;
+ return first.get_bool();
}
simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator rvv_vls::ondemand::value() noexcept(false) {
if (error()) { throw simdjson_error(error()); }
@@ -181574,6 +233598,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondeman
/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
#include <algorithm>
+#include <cstring>
#include <stdexcept>
namespace simdjson {
@@ -181660,23 +233685,20 @@ simdjson_inline document_stream::document_stream(
const uint8_t *_buf,
size_t _len,
size_t _batch_size,
- bool _allow_comma_separated
+ bool _allow_comma_separated,
+ stream_format _format
) noexcept
: parser{&_parser},
buf{_buf},
len{_len},
batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
allow_comma_separated{_allow_comma_separated},
+ format{_format},
error{SUCCESS}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
#endif
{
-#ifdef SIMDJSON_THREADS_ENABLED
- if(worker.get() == nullptr) {
- error = MEMALLOC;
- }
-#endif
}
simdjson_inline document_stream::document_stream() noexcept
@@ -181685,6 +233707,7 @@ simdjson_inline document_stream::document_stream() noexcept
len{0},
batch_size{0},
allow_comma_separated{false},
+ format{stream_format::whitespace_delimited},
error{UNINITIALIZED}
#ifdef SIMDJSON_THREADS_ENABLED
, use_thread(false)
@@ -181704,6 +233727,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
}
inline size_t document_stream::truncated_bytes() const noexcept {
+ // Stage 1 returns EMPTY on zero-length input before it writes the index
+ // sentinels read below, so they would still hold a previous stream's values.
+ if (len == 0) { return 0; }
if(error == CAPACITY) { return len - batch_start; }
return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
}
@@ -181784,13 +233810,20 @@ inline void document_stream::start() noexcept {
error = run_stage1(*parser, batch_start);
}
if (error) { return; }
- doc_index = batch_start;
+ // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+ // after the RS delimiter and any following whitespace. For regular mode, it is
+ // the offset from batch_start to the first document in the batch.
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
doc = document(json_iterator(&buf[batch_start], parser));
doc.iter._streaming = true;
#ifdef SIMDJSON_THREADS_ENABLED
if (use_thread && next_batch_start() < len) {
// Kick off the first thread on next batch if needed
+ if (worker.get() == nullptr) {
+ worker.reset(new(std::nothrow) stage1_worker());
+ if (worker.get() == nullptr) { error = MEMALLOC; return; }
+ }
error = stage1_thread_parser.allocate(batch_size);
if (error) { return; }
worker->start_thread();
@@ -181865,12 +233898,69 @@ inline void document_stream::next() noexcept {
*/
if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
- doc_index = batch_start;
+ doc_index = batch_start + parser->implementation->structural_indexes[0];
}
}
}
+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+ switch (format) {
+ case stream_format::newline_delimited: return '\n';
+ case stream_format::json_sequence: return 0x1E;
+ default: return 0;
+ }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+ const uint8_t *const base = &buf[batch_start];
+ const token_position pos = doc.iter.position();
+ const token_position end = doc.iter.end_position();
+ if (pos >= end) { return false; }
+ const size_t here = size_t(doc.iter.token.peek(pos) - base);
+ const size_t batch_len =
+ (len - batch_start < batch_size) ? len - batch_start : batch_size;
+ if (here >= batch_len) { return false; }
+ const uint8_t *const found = static_cast<const uint8_t *>(
+ std::memchr(base + here, delimiter, batch_len - here));
+ if (found == nullptr) { return false; }
+
+ const uint32_t boundary = uint32_t(found - base);
+ // The answer is near `pos`: the delimiter ends the current document, while
+ // `end` spans the whole batch. Gallop first so the cost follows the distance
+ // rather than the size of the batch.
+ token_position lo = pos;
+ size_t hop = 1;
+ while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+ token_position hi = (lo + hop < end) ? lo + hop : end;
+ while (lo < hi) {
+ const token_position mid = lo + ((hi - lo) >> 1);
+ if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+ }
+ doc.iter.token.set_position(lo);
+ return true;
+}
+
inline void document_stream::next_document() noexcept {
+ // A delimiter that cannot occur inside a document tells us where the current
+ // one ends, so we can jump there instead of walking every structural. Only
+ // valid while the iterator is still inside the document: a consumed document
+ // already sits on the next one's first token, and skip_child() returns at
+ // once for it.
+ //
+ // The jump does not structure-validate the unread remainder of the current
+ // document: under newline_delimited / json_sequence the next delimiter is
+ // assumed to be the true document boundary. Callers that leave depth() > 0
+ // while violating that contract (e.g. pretty multi-line JSON under
+ // newline_delimited) can mis-align following documents; use
+ // whitespace_delimited if unsure.
+ const uint8_t delimiter = document_delimiter();
+ if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+ skip_to_delimiter(delimiter)) {
+ doc.iter._depth = 1;
+ doc.iter._string_buf_loc = parser->string_buf.get();
+ doc.iter._root = doc.iter.position();
+ return;
+ }
// Go to next place where depth=0 (document depth)
error = doc.iter.skip_child(0);
if (error) { return; }
@@ -181894,10 +233984,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
// This code only updates the structural index in the parser, it does not update any json_iterator
// instance.
size_t remaining = len - _batch_start;
+ stage1_mode mode;
if (remaining <= batch_size) {
- return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+ // Final batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_final;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_final;
+ break;
+ default:
+ mode = stage1_mode::streaming_final;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], remaining, mode);
} else {
- return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+ // Partial batch
+ switch (format) {
+ case stream_format::json_sequence:
+ mode = stage1_mode::json_sequence_partial;
+ break;
+ case stream_format::comma_delimited:
+ mode = stage1_mode::comma_delimited_partial;
+ break;
+ default:
+ mode = stage1_mode::streaming_partial;
+ break;
+ }
+ return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
}
}
@@ -181906,11 +234021,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
}
simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
- auto depth = stream->doc.iter.depth();
+ // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+ // the input, as the DOM document_stream does.
+ if (stream->error) {
+ return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+ }
+ // Always walk from the root of the document, whatever the current position
+ // of the document iterator: the user may have already consumed part of the
+ // document, so the iterator's current depth must not be used here.
+ depth_t depth = 1;
auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();
- // If at root, process the first token to determine if scalar value
- if (stream->doc.iter.at_root()) {
+ // Process the first token to determine if scalar value
+ {
switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
case '{': case '[': // Depth=1 already at start of document
break;
@@ -181918,14 +234041,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
depth--;
break;
default: // Scalar value document
- // TODO: We could remove trailing whitespaces
// This returns a string spanning from start of value to the beginning of the next document (excluded)
{
auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
// normally the length would be next_index - current_index() - 1, except for the last document
size_t svlen = next_index - current_index();
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
- while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+ // When the scalar is followed by a truncated document, the structural
+ // indexes of that document were dropped and next_index is the end of
+ // the input, so we bound the scalar by scanning the token itself.
+ size_t token_len = 0;
+ if (*start == '"') {
+ token_len = 1;
+ while (token_len < svlen) {
+ char c = start[token_len++];
+ if (c == '\\') {
+ token_len++;
+ } else if (c == '"') {
+ break;
+ }
+ }
+ } else {
+ while (token_len < svlen) {
+ char c = start[token_len];
+ if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+ break;
+ }
+ token_len++;
+ }
+ }
+ if (token_len > 0 && token_len < svlen) {
+ svlen = token_len;
+ }
+ // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+ // json_sequence mode the scanner classifies RS as a scalar
+ // character, so an RS-prefixed scalar document (number / true /
+ // false / null / string) has no closing structural index and the
+ // slice runs all the way up to the next document's RS. RS cannot
+ // legally appear in a JSON value at the source level (control
+ // characters in strings must be escaped as \u001E), so stripping
+ // it is safe in every stream_format.
+ while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
svlen--;
}
return std::string_view(start, svlen);
@@ -182050,11 +234206,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
return answer;
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+ std::string_view key;
+ SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+ return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template <typename string_type>
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
std::string_view key;
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
- receiver = key;
+ internal::assign_utf8(receiver, key);
return SUCCESS;
}
@@ -182076,6 +234240,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+ return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline value &field::value() & noexcept {
return second;
}
@@ -182120,11 +234290,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondem
return first.escaped_key();
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::field>::escaped_u8key() noexcept {
+ if (error()) { return error(); }
+ return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
if (error()) { return error(); }
return first.unescaped_key(allow_replacement);
}
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+ if (error()) { return error(); }
+ return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
template<typename string_type>
simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
if (error()) { return error(); }
@@ -182168,6 +234352,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
_depth{other._depth},
_root{other._root},
_streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
{
other.parser = nullptr;
}
@@ -182179,6 +234366,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
_depth = other._depth;
_root = other._root;
_streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ _allow_incomplete_json = other._allow_incomplete_json;
+#endif
other.parser = nullptr;
return *this;
}
@@ -182205,7 +234395,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
_string_buf_loc{parser->string_buf.get()},
_depth{1},
_root{parser->implementation->structural_indexes.get()},
- _streaming{streaming}
+ _streaming{streaming},
+ _allow_incomplete_json{true}
{
logger::log_headers();
@@ -182277,7 +234468,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
#endif // SIMDJSON_CHECK_EOF
break;
case '"':
- if(*peek() == ':') {
+ // At the end, peek() would read the sentinel, which points into the padding.
+ if(!at_end() && *peek() == ':') {
// We are at a key!!!
// This might happen if you just started an object and you skip it immediately.
// Performance note: it would be nice to get rid of this check as it is somewhat
@@ -182320,7 +234512,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
}
}
- return report_error(TAPE_ERROR, "not enough close braces");
+ return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
}
SIMDJSON_POP_DISABLE_WARNINGS
@@ -182337,6 +234529,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
return _streaming;
}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+ return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+ const uint8_t *end = token.buf + parser->_document_len;
+ return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline token_position json_iterator::root_position() const noexcept {
return _root;
}
@@ -182619,7 +234822,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
case json_type::string: out << "string"; break;
case json_type::boolean: out << "boolean"; break;
case json_type::null: out << "null"; break;
- default: SIMDJSON_UNREACHABLE();
+ case json_type::unknown: out << "unknown"; break;
}
return out;
}
@@ -182958,6 +235161,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
/* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
/* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string */
/* amalgamation skipped (editor-only): #include <meta> */
@@ -182987,12 +235194,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return std::forward<object>(*this).find_field_unordered(key);
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -183002,6 +235218,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool has_value;
SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
if (!has_value) {
@@ -183011,6 +235230,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
return value(iter.child());
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+ // Single pass driven directly by the value_iterator, mirroring
+ // find_field_unordered_raw + value(iter.child()). Compared to walking via
+ // object_iterator/field, this avoids constructing a simdjson_result<field> and
+ // a field (key + value) for every field -- and the development-check bookkeeping
+ // in object_iterator -- building a value only for the fields that actually match.
+ // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // Mirror object::begin(): for_each must start at the beginning of the object,
+ // not from some position left behind by a prior find_field on the same object.
+ if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+ value_iterator it = iter;
+ std::size_t matched = 0;
+ // Track which selector indices have already matched, as a compile-time bitset
+ // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+ // occurrence, and we stop as soon as every key has matched.
+ constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+ std::array<std::uint64_t, seen_words> seen{};
+ while (it.is_open()) {
+ raw_json_string key;
+ error_code error;
+ std::size_t idx;
+ if constexpr (Selector::window.ok) {
+ // A window selector confirms a key from its raw bytes alone (the closing
+ // quote bounds it), so we take the length-free path: field_key (no backward
+ // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+ if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key);
+ } else {
+ // Otherwise derive the key length from the structural index (the following
+ // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+ std::size_t key_len;
+ if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+ if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+ idx = Selector::match_raw(key.raw(), key_len);
+ }
+ if (idx < Selector::size()) {
+ const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+ std::uint64_t &seen_word = seen[idx >> 6];
+ if (!(seen_word & seen_bit)) {
+ seen_word |= seen_bit;
+ value matched_value(it.child());
+ // The callback may return void or anything convertible to error_code
+ // (error_code itself, or a for_each_result from a nested for_each). When
+ // it yields an error_code, we stop at the first non-SUCCESS result and
+ // propagate it so the caller can surface value-parse errors (for example,
+ // a type mismatch on a matched field). A void-returning callback is
+ // responsible for handling its own errors.
+ if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+ // Unlike the internal-error paths above, a callback error does not
+ // abandon the iterator: we leave it recoverable so the caller can keep
+ // using the object (or its parent) after handling the error.
+ if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+ } else {
+ on_match(idx, matched_value);
+ }
+ if (++matched >= Selector::size()) { break; }
+ }
+ }
+ // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+ // the value and left the iterator closed or in error (e.g. a void callback
+ // that swallowed a fatal sub-iteration error), stop here rather than calling
+ // skip_child on a closed iterator.
+ if (!it.is_open()) { break; }
+ // Skip the value (a no-op if the callback consumed it) and step to the next
+ // field; has_next_field() ends the container on '}', which closes the loop.
+ if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+ if ((error = it.has_next_field().error())) { return {error, matched}; }
+ }
+ return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+ std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+ error_code err = SUCCESS;
+ auto try_one = [&](auto Ic) {
+ constexpr std::size_t I = decltype(Ic)::value;
+ if (idx == I) {
+ auto&& h = std::get<I>(handlers);
+ using H = std::remove_reference_t<decltype(h)>;
+ if constexpr (std::is_invocable_v<H&, value>) {
+ // A handler returning void runs for its side effects; one returning
+ // anything convertible to error_code (error_code, or a for_each_result
+ // from a nested for_each) has its error captured and propagated.
+ if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+ err = h(v);
+ } else {
+ h(v);
+ }
+ } else {
+ // Direct deserialization target: assign the matched value into it.
+ err = v.get(h);
+ }
+ }
+ };
+ (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+ return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+ requires key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+ // Reuse the single (index, value) implementation via a tiny adapter.
+ // The adapter is called once per *matched* key (very few); the hot path
+ // (iteration + match_raw + seen bitset) stays exactly the same.
+ return this->template for_each<Selector>(
+ [&](std::size_t i, value v) -> error_code {
+ return key_selector_for_each_detail::dispatch_value(
+ i, handlers, v, std::make_index_sequence<Selector::size()>{});
+ });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+ noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ using Selector = key_selector<Keys...>;
+ return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
SIMDJSON_TRY( iter.start_object().error() );
return object(iter);
@@ -183046,6 +235409,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
}
simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
const uint8_t * starting_point{iter.peek_start()};
auto error = consume();
if(error) { return error; }
@@ -183067,9 +235433,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
{
}
-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
- if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+ return object_iterator(iter, this);
+#endif
+ return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ // The object is a temporary that the iterator may outlive: do not lock it.
+ if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
#endif
return object_iterator(iter);
}
@@ -183078,7 +235452,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
}
inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
- if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+ // An empty pointer has no json_pointer[0]: with a default-constructed
+ // std::string_view, reading it dereferences a null pointer.
+ if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
json_pointer = json_pointer.substr(1);
size_t slash = json_pointer.find('/');
std::string_view key = json_pointer.substr(0, slash);
@@ -183180,6 +235556,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
}
simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
bool is_not_empty;
auto error = iter.reset_object().get(is_not_empty);
if(error) { return error; }
@@ -183187,9 +235566,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
}
simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
return iter.reset_object();
}
+simdjson_inline object_position object::get_current_position() const noexcept {
+ return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+ if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+ // json_iterator::reenter_child() requires the live depth to be exactly
+ // one level shallower than the target (matching how every other depth
+ // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+ // additionally validates against the parser's per-depth container-start
+ // bookkeeping. Neither applies here: depending on what was captured and
+ // what has happened since (a scalar field fully consumed, a compound
+ // value left open, a find_field() miss that scanned past everything),
+ // the live depth when reverting can be any number of levels away from
+ // the captured one, and the captured depth is not necessarily a
+ // container's own start. reenter_at() moves directly, matching how
+ // reset_object() itself repositions without going through reenter_child().
+ iter.reenter_at(position.position, position.depth);
+ return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+ locked = _locked;
+}
+#endif
+
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
template<constevalutil::fixed_string... FieldNames, typename T>
@@ -183247,10 +235658,14 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::object>::simdjson_result(rvv_
simdjson_inline simdjson_result<rvv_vls::ondemand::object>::simdjson_result(error_code error) noexcept
: implementation_simdjson_result_base<rvv_vls::ondemand::object>(error) {}
-simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() & noexcept {
if (error()) { return error(); }
return first.begin();
}
+simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() && noexcept {
+ if (error()) { return error(); }
+ return std::move(first).begin();
+}
simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::end() noexcept {
if (error()) { return error(); }
return first.end();
@@ -183304,11 +235719,55 @@ simdjson_inline error_code simdjson_result<rvv_vls::ondemand::object>::for_each_
return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
}
+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+ requires rvv_vls::ondemand::key_selector_type<Selector> &&
+ std::is_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Func&& on_match)
+ noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+ requires rvv_vls::ondemand::key_selector_type<Selector> &&
+ (sizeof...(Handlers) == Selector::size()) &&
+ (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+ requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+ (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+ (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Handlers&&... on_match)
+ noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+ if (error()) { return {error(), 0}; }
+ return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::object>::reset() noexcept {
if (error()) { return error(); }
return first.reset();
}
+inline simdjson_result<rvv_vls::ondemand::object_position> simdjson_result<rvv_vls::ondemand::object>::get_current_position() noexcept {
+ if (error()) { return error(); }
+ return first.get_current_position();
+}
+
+inline error_code simdjson_result<rvv_vls::ondemand::object>::revert_position(rvv_vls::ondemand::object_position position) noexcept {
+ if (error()) { return error(); }
+ return first.revert_position(position);
+}
+
inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::object>::is_empty() noexcept {
if (error()) { return error(); }
return first.is_empty();
@@ -183352,6 +235811,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
: iter{_iter}
{}
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+ : parent{_parent}, iter{_iter}
+{
+ if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+ if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{other.parent},
+ iter{std::move(other.iter)}
+{
+ other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = other.parent;
+ iter = std::move(other.iter);
+
+ other.parent = nullptr;
+ }
+ return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+ : has_been_referenced{other.has_been_referenced},
+ parent{nullptr},
+ iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+ if (this != &other)
+ {
+ if (parent)
+ parent->set_locked(false);
+ has_been_referenced = other.has_been_referenced;
+ parent = nullptr;
+ iter = other.iter;
+ }
+ return *this;
+}
+#endif
+
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
#if SIMDJSON_DEVELOPMENT_CHECKS
// We must call * once per iteration.
@@ -183479,6 +235993,147 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> &simdjson_re
#endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
/* end file simdjson/generic/ondemand/object_iterator-inl.h for rvv_vls */
+/* including simdjson/generic/ondemand/ranges-inl.h for rvv_vls: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+ auto b = arr.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+ return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+ return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+ : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+ return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+ ++iter_;
+ return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+ ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+ auto b = obj.begin();
+ if (b.error()) { error_ = b.error(); return; }
+ begin_ = b.value_unsafe();
+ end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+ return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+ return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+ return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+ return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+ return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+ return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::rvv_vls::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::rvv_vls::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::rvv_vls::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::rvv_vls::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::rvv_vls::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::rvv_vls::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for rvv_vls */
/* including simdjson/generic/ondemand/parser-inl.h for rvv_vls: #include "simdjson/generic/ondemand/parser-inl.h" */
/* begin file simdjson/generic/ondemand/parser-inl.h for rvv_vls */
#ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -183510,7 +236165,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
// string_capacity copied from document::allocate
_capacity = 0;
- size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+ if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+ return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+ }
+ size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
#if SIMDJSON_DEVELOPMENT_CHECKS
start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -183535,6 +236193,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -183551,6 +236210,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
json.remove_utf8_bom();
+ _document_len = json.length();
// Allocate if needed
if (capacity() < json.length() || !string_buf) {
@@ -183616,6 +236276,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
}
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+ return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+ return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
// Warning: no check is done on the buffer padding. We trust the user.
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -183623,8 +236311,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
buf += 3;
len -= 3;
}
- if(allow_comma_separated && batch_size < len) { batch_size = len; }
- return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+ // Map allow_comma_separated to stream_format::comma_delimited
+ if (allow_comma_separated) {
+ return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+ }
+ return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
}
inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -183644,6 +236335,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
return iterate_many(pad(s), batch_size, allow_comma_separated);
}
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+ if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+ buf += 3;
+ len -= 3;
+ }
+ if (format == stream_format::comma_delimited_array) {
+ // Strip leading JSON whitespace.
+ while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+ buf++; len--;
+ }
+ // Expect the opening '['.
+ if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+ buf++; len--;
+ // Strip trailing JSON whitespace.
+ while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+ len--;
+ }
+ // Expect the closing ']'.
+ if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+ len--;
+ // Fall through to comma_delimited over the array contents.
+ format = stream_format::comma_delimited;
+ }
+ return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+ if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+ return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+ return iterate_many(padded_string_view(s), batch_size, format);
+}
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
return _capacity;
}
@@ -184051,6 +236787,27 @@ namespace simdjson {
namespace rvv_vls {
namespace ondemand {
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+ bool escaping{false};
+ for (uint32_t i = 1; i < max_len; i++) {
+ switch (json[i]) {
+ case '"':
+ if (!escaping) { return true; }
+ escaping = false;
+ break;
+ case '\\':
+ escaping = !escaping;
+ break;
+ default:
+ escaping = false;
+ break;
+ }
+ }
+ return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
simdjson_inline value_iterator::value_iterator(
json_iterator *json_iter,
depth_t depth,
@@ -184438,6 +237195,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
return raw_json_string(key);
}
+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+ assert_at_next();
+
+ const uint8_t *k = _json_iter->return_current_and_advance();
+ if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+ // After return_current_and_advance(), the current token is the ':' that follows
+ // the key. The closing quote sits just before it (only JSON whitespace may
+ // intervene), so step back from the ':' to the closing quote to get the length.
+ // In minified JSON this is a single back-step.
+ const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+ do { --q; } while (*q != '"');
+ key = raw_json_string(k);
+ len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+ return SUCCESS;
+}
+
simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
assert_at_next();
@@ -184555,7 +237328,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_string(allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -184566,6 +237339,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -184599,6 +237381,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+ auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+ auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+ if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+ return result;
+}
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
auto result = parse_bool(peek_non_root_scalar("bool"));
if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -184701,7 +237493,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
auto saved_string_buf_loc = _json_iter->string_buf_loc();
auto err = get_root_string(check_trailing, allow_replacement).get(content);
if (err) { return err; }
- receiver = content;
+ internal::assign_utf8(receiver, content);
// Restore the string buffer location, effectively discarding any temporary string storage
_json_iter->string_buf_loc() = saved_string_buf_loc;
return SUCCESS;
@@ -184713,6 +237505,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
auto json = peek_scalar("string");
if (*json != '"') { return incorrect_type_error("Not a string"); }
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+ if (_json_iter->allow_incomplete_json()) {
+ const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+ const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+ if (!raw_json_string_is_quote_terminated(json, max_len)) {
+ return STRING_ERROR;
+ }
+ }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
advance_scalar("string");
return raw_json_string(json+1);
}
@@ -184822,6 +237623,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
return result;
}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ // We use the same buffer size as get_root_double: the number of significant
+ // digits that matter is smaller for binary32, but the JSON document may still
+ // spell out a long number that we must parse (and round) faithfully.
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+ auto max_len = peek_root_length();
+ auto json = peek_root_scalar("float");
+ uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+ tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+ if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+ logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+ return NUMBER_ERROR;
+ }
+ auto result = numberparsing::parse_float_in_string(tmpbuf);
+ if(result.error() == SUCCESS) {
+ if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+ advance_root_scalar("float");
+ }
+ return result;
+}
+
simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
auto max_len = peek_root_length();
auto json = peek_root_scalar("bool");
@@ -185050,6 +237888,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
_json_iter->token.set_position(_start_position + 1);
}
+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+ // Unlike reenter_child(), this does not require the live depth to be
+ // exactly one level shallower than depth, nor does it validate against
+ // the parser's per-depth container-start bookkeeping: neither holds in
+ // general for a caller-supplied snapshot (see object_position). What
+ // must still always hold, regardless of what was captured or how far
+ // the live iterator has since moved, is that position and depth are
+ // themselves sane values -- this is the same bound reenter_child()
+ // itself applies unconditionally.
+ SIMDJSON_ASSUME(position != nullptr);
+ SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+ _json_iter->_depth = depth;
+ _json_iter->token.set_position(position);
+}
+
simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
if(error()) { return error(); }
move_at_container_start();
@@ -186430,12 +239283,12 @@ public:
explicit auto_parser(std::remove_pointer_t<parser_type> &parser, padded_string_view const str) noexcept requires(std::is_pointer_v<parser_type>);
explicit auto_parser(padded_string_view const str) noexcept requires(std::is_pointer_v<parser_type>);
explicit auto_parser(parser_type parser, ondemand::document &&doc) noexcept requires(std::is_pointer_v<parser_type>);
- auto_parser(auto_parser const &) = delete;
- auto_parser &operator=(auto_parser const &) = delete;
- auto_parser(auto_parser &&) noexcept = default;
- auto_parser &operator=(auto_parser &&) noexcept = default;
- ~auto_parser() = default;
-
+ auto_parser(auto_parser const &) = delete;
+ auto_parser &operator=(auto_parser const &) = delete;
+ ~auto_parser() = default;
+ // Prevent moving
+ auto_parser(auto_parser&&) = delete;
+ auto_parser &operator=(auto_parser &&) noexcept = delete;
simdjson_warn_unused std::remove_pointer_t<parser_type> &parser() noexcept;
template <typename T>
@@ -186469,11 +239322,6 @@ struct to_adaptor {
T operator()(simdjson_result<ondemand::value> &val) const noexcept;
auto operator()(padded_string_view const str) const noexcept;
auto operator()(ondemand::parser &parser, padded_string_view const str) const noexcept;
- // The std::string is padded with reserve to ensure there is enough space for padding.
- // Some sanitizers may not like this, so you can use simdjson::pad instead.
- // simdjson::from(simdjson::pad(str))
- auto operator()(std::string str) const noexcept;
- auto operator()(ondemand::parser &parser, std::string str) const noexcept;
};
// deduction guide
auto_parser(padded_string_view const str) -> auto_parser<ondemand::parser*>;
@@ -186484,7 +239332,12 @@ auto_parser(padded_string_view const str) -> auto_parser<ondemand::parser*>;
* The simdjson::from instance is EXPERIMENTAL AND SUBJECT TO CHANGES.
*
* The `from` instance is a utility adaptor for parsing JSON strings into objects.
- * It provides a convenient way to convert JSON data into C++ objects using the `auto_parser`.
+ *
+ * The string must be a simdjson::padded_string_view, which can be created from a std::string
+ * with simdjson::pad(), from a simdjson::padded_string, or string literal using the `_padded`
+ * user-defined literal.
+ *
+ * The `from` instance provides a convenient way to convert JSON data into C++ objects using the `auto_parser`.
*
* Example usage:
*
@@ -186555,9 +239408,6 @@ inline auto_parser<parser_type>::auto_parser(parser_type parser, ondemand::docum
: auto_parser{*parser, std::move(doc)} {}
-
-
-
template <typename parser_type>
inline std::remove_pointer_t<parser_type> &auto_parser<parser_type>::parser() noexcept {
if constexpr (std::is_pointer_v<parser_type>) {
@@ -186636,16 +239486,6 @@ template <typename T>
inline auto to_adaptor<T>::operator()(ondemand::parser &parser, padded_string_view const str) const noexcept {
return auto_parser<ondemand::parser *>{parser, str};
}
-
-template <typename T>
-inline auto to_adaptor<T>::operator()(std::string str) const noexcept {
- return auto_parser<ondemand::parser *>{pad_with_reserve(str)};
-}
-
-template <typename T>
-inline auto to_adaptor<T>::operator()(ondemand::parser &parser, std::string str) const noexcept {
- return auto_parser<ondemand::parser *>{parser, pad_with_reserve(str)};
-}
} // namespace internal
} // namespace convert
} // namespace simdjson
@@ -186721,14 +239561,17 @@ namespace compile_time {
template <constevalutil::fixed_string json_str> consteval auto parse_json();
} // namespace compile_time
-} // namespace simdjson
+inline namespace literals {
template <simdjson::constevalutil::fixed_string str>
consteval auto operator ""_json() {
return simdjson::compile_time::parse_json<str>();
}
+} // namespace literals
+} // namespace simdjson
+
#endif // SIMDJSON_STATIC_REFLECTION
#endif // SIMDJSON_GENERIC_COMPILE_TIME_JSON_H
/* end file simdjson/compile_time_json.h */
@@ -186748,42 +239591,649 @@ consteval auto operator ""_json() {
#if SIMDJSON_STATIC_REFLECTION
/* skipped duplicate #include "simdjson/compile_time_json.h" */
-#include <array>
-#include <cstdint>
-#include <meta>
-#include <string_view>
+/* including simdjson/internal/fast_float.h: #include "simdjson/internal/fast_float.h" */
+/* begin file simdjson/internal/fast_float.h */
+// Vendored from fast_float v8.2.10, generated by tools/vendor_fast_float.sh.
+// Do not edit by hand; re-run the script to update.
+//
+// https://github.com/fastfloat/fast_float
+// Licensed under Apache-2.0 OR MIT OR BSL-1.0, at your option.
+//
+// simdjson uses this for two things that its own number parser cannot do:
+// * the slow path for numbers with more than 19 significant digits, where
+// fast_float's bigint comparison is several times quicker than the
+// Wuffs-derived decimal shifting it replaced (see src/from_chars.cpp), and
+// * correctly rounded parsing inside a constant expression, which the runtime
+// path cannot offer because it relies on memcpy and __uint128_t (see
+// compile_time_json-inl.h).
+//
+// Two edits are applied by the script. Every fast_float name is rewritten so
+// that this copy cannot collide with a copy of fast_float that the surrounding
+// program includes for itself: namespace fast_float -> simdjson_fast_float,
+// FASTFLOAT_* -> SIMDJSON_FASTFLOAT_*, fastfloat_* -> simdjson_fastfloat_*. And
+// the accented letters in the attribution comments below are folded to ASCII,
+// to keep the tree ASCII-only; no disrespect to the people named is intended.
+// simdjson_fast_float by Daniel Lemire
+// simdjson_fast_float by Joao Paulo Magalhaes
+//
+//
+// with contributions from Eugene Golushkov
+// with contributions from Maksim Kita
+// with contributions from Marcin Wojdyr
+// with contributions from Neal Richardson
+// with contributions from Tim Paine
+// with contributions from Fabio Pellacini
+// with contributions from Lenard Szolnoki
+// with contributions from Jan Pharago
+// with contributions from Maya Warrier
+// with contributions from Taha Khokhar
+// with contributions from Anders Dalvander
+//
+//
+// Licensed under the Apache License, Version 2.0, or the
+// MIT License or the Boost License. This file may not be copied,
+// modified, or distributed except according to those terms.
+//
+// MIT License Notice
+//
+// MIT License
+//
+// Copyright (c) 2021 The simdjson_fast_float authors
+//
+// Permission is hereby granted, free of charge, to any
+// person obtaining a copy of this software and associated
+// documentation files (the "Software"), to deal in the
+// Software without restriction, including without
+// limitation the rights to use, copy, modify, merge,
+// publish, distribute, sublicense, and/or sell copies of
+// the Software, and to permit persons to whom the Software
+// is furnished to do so, subject to the following
+// conditions:
+//
+// The above copyright notice and this permission notice
+// shall be included in all copies or substantial portions
+// of the Software.
+//
+// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF
+// ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED
+// TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
+// PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT
+// SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
+// CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
+// OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR
+// IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+// DEALINGS IN THE SOFTWARE.
+//
+// Apache License (Version 2.0) Notice
+//
+// Copyright 2021 The simdjson_fast_float authors
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+//
+// BOOST License Notice
+//
+// Boost Software License - Version 1.0 - August 17th, 2003
+//
+// Permission is hereby granted, free of charge, to any person or organization
+// obtaining a copy of the software and accompanying documentation covered by
+// this license (the "Software") to use, reproduce, display, distribute,
+// execute, and transmit the Software, and to prepare derivative works of the
+// Software, and to permit third-parties to whom the Software is furnished to
+// do so, all subject to the following:
+//
+// The copyright notices in the Software and this entire statement, including
+// the above license grant, this restriction and the following disclaimer,
+// must be included in all copies of the Software, in whole or in part, and
+// all derivative works of the Software, unless such copies or derivative
+// works are solely in the form of machine-executable object code generated by
+// a source language processor.
+//
+// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+// FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
+// SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
+// FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
+// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+// DEALINGS IN THE SOFTWARE.
+//
-#include <algorithm>
-#include <array>
-#include <charconv>
+#ifndef SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+#define SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifdef __has_include
+#if __has_include(<version>)
+#include <version>
+#endif
+#endif
+
+// Testing for https://wg21.link/N3652, adopted in C++14
+#if defined(__cpp_constexpr) && __cpp_constexpr >= 201304
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14 constexpr
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14
+#endif
+
+#if defined(__cpp_lib_bit_cast) && __cpp_lib_bit_cast >= 201806L
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 0
+#endif
+
+#if defined(__cpp_lib_is_constant_evaluated) && \
+ __cpp_lib_is_constant_evaluated >= 201811L
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 0
+#endif
+
+#if defined(__cpp_if_constexpr) && __cpp_if_constexpr >= 201606L
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if constexpr (x)
+#else
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if (x)
+#endif
+
+// Testing for relevant C++20 constexpr library features
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST && \
+ defined(__cpp_lib_constexpr_algorithms) && \
+ __cpp_lib_constexpr_algorithms >= 201806L /*For std::copy and std::fill*/
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20 constexpr
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 1
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 0
+#endif
+
+#if __cplusplus >= 201703L || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L)
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 0
+#else
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 1
+#endif
+
+#endif // SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifndef SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+#define SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+
+#include <cfloat>
+#include <cstddef>
#include <cstdint>
-#include <expected>
-#include <meta>
-#include <string>
-#include <string_view>
-#include <vector>
+#include <cassert>
+#include <cstring>
+#include <limits>
+#include <type_traits>
+#include <system_error>
+#ifdef __has_include
+#if __has_include(<stdfloat>) && (__cplusplus > 202002L || (defined(_MSVC_LANG) && (_MSVC_LANG > 202002L)))
+#include <stdfloat>
+#endif
+#endif
-#define simdjson_consteval_error(...) \
+#define SIMDJSON_FASTFLOAT_VERSION_MAJOR 8
+#define SIMDJSON_FASTFLOAT_VERSION_MINOR 2
+#define SIMDJSON_FASTFLOAT_VERSION_PATCH 10
+
+#define SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x) #x
+#define SIMDJSON_FASTFLOAT_STRINGIZE(x) SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x)
+
+#define SIMDJSON_FASTFLOAT_VERSION_STR \
+ SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MAJOR) \
+ "." SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MINOR) "." SIMDJSON_FASTFLOAT_STRINGIZE( \
+ SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+#define SIMDJSON_FASTFLOAT_VERSION \
+ (SIMDJSON_FASTFLOAT_VERSION_MAJOR * 10000 + SIMDJSON_FASTFLOAT_VERSION_MINOR * 100 + \
+ SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+namespace simdjson_fast_float {
+
+enum class chars_format : uint64_t;
+
+namespace detail {
+constexpr chars_format basic_json_fmt = chars_format(1 << 5);
+constexpr chars_format basic_fortran_fmt = chars_format(1 << 6);
+} // namespace detail
+
+enum class chars_format : uint64_t {
+ scientific = 1 << 0,
+ fixed = 1 << 2,
+ hex = 1 << 3,
+ no_infnan = 1 << 4,
+ // RFC 8259: https://datatracker.ietf.org/doc/html/rfc8259#section-6
+ json = uint64_t(detail::basic_json_fmt) | fixed | scientific | no_infnan,
+ // Extension of RFC 8259 where, e.g., "inf" and "nan" are allowed.
+ json_or_infnan = uint64_t(detail::basic_json_fmt) | fixed | scientific,
+ fortran = uint64_t(detail::basic_fortran_fmt) | fixed | scientific,
+ general = fixed | scientific,
+ allow_leading_plus = 1 << 7,
+ skip_white_space = 1 << 8,
+};
+
+template <typename UC> struct from_chars_result_t {
+ UC const *ptr;
+ std::errc ec;
+
+ // https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p2497r0.html
+ constexpr explicit operator bool() const noexcept {
+ return ec == std::errc();
+ }
+};
+
+using from_chars_result = from_chars_result_t<char>;
+
+template <typename UC> struct parse_options_t {
+ constexpr explicit parse_options_t(chars_format fmt = chars_format::general,
+ UC dot = UC('.'), int b = 10)
+ : format(fmt), decimal_point(dot), base(b) {}
+
+ /** Which number formats are accepted */
+ chars_format format;
+ /** The character used as decimal point */
+ UC decimal_point;
+ /** The base used for integers */
+ int base;
+};
+
+using parse_options = parse_options_t<char>;
+
+} // namespace simdjson_fast_float
+
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+#include <bit>
+#endif
+
+#if (defined(__x86_64) || defined(__x86_64__) || defined(_M_X64) || \
+ defined(__amd64) || defined(__aarch64__) || defined(_M_ARM64) || \
+ defined(__MINGW64__) || defined(__s390x__) || \
+ (defined(__ppc64__) || defined(__PPC64__) || defined(__ppc64le__) || \
+ defined(__PPC64LE__)) || \
+ defined(__loongarch64) || (defined(__riscv) && __riscv_xlen == 64))
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#elif (defined(__i386) || defined(__i386__) || defined(_M_IX86) || \
+ defined(__arm__) || defined(_M_ARM) || defined(__ppc__) || \
+ defined(__MINGW32__) || defined(__EMSCRIPTEN__) || \
+ (defined(__riscv) && __riscv_xlen == 32))
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#else
+ // Need to check incrementally, since SIZE_MAX is a size_t, avoid overflow.
+// We can never tell the register width, but the SIZE_MAX is a good
+// approximation. UINTPTR_MAX and INTPTR_MAX are optional, so avoid them for max
+// portability.
+#if SIZE_MAX == 0xffff
+#error Unknown platform (16-bit, unsupported)
+#elif SIZE_MAX == 0xffffffff
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#elif SIZE_MAX == 0xffffffffffffffff
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#else
+#error Unknown platform (not 32-bit, not 64-bit?)
+#endif
+#endif
+
+#if ((defined(_WIN32) || defined(_WIN64)) && !defined(__clang__)) || \
+ (defined(_M_ARM64) && !defined(__MINGW32__))
+#include <intrin.h>
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#define SIMDJSON_FASTFLOAT_VISUAL_STUDIO 1
+#endif
+
+#if defined __BYTE_ORDER__ && defined __ORDER_BIG_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+#elif defined _WIN32
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#if defined(__APPLE__) || defined(__FreeBSD__)
+#include <machine/endian.h>
+#elif defined(sun) || defined(__sun)
+#include <sys/byteorder.h>
+#elif defined(__MVS__)
+#include <sys/endian.h>
+#else
+#ifdef __has_include
+#if __has_include(<endian.h>)
+#include <endian.h>
+#endif //__has_include(<endian.h>)
+#endif //__has_include
+#endif
+#
+#ifndef __BYTE_ORDER__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#ifndef __ORDER_LITTLE_ENDIAN__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 1
+#endif
+#endif
+
+#if defined(__SSE2__) || (defined(SIMDJSON_FASTFLOAT_VISUAL_STUDIO) && \
+ (defined(_M_AMD64) || defined(_M_X64) || \
+ (defined(_M_IX86_FP) && _M_IX86_FP == 2)))
+#define SIMDJSON_FASTFLOAT_SSE2 1
+#endif
+
+#if defined(__aarch64__) || defined(_M_ARM64)
+#define SIMDJSON_FASTFLOAT_NEON 1
+#endif
+
+#if defined(SIMDJSON_FASTFLOAT_SSE2) || defined(SIMDJSON_FASTFLOAT_NEON)
+#define SIMDJSON_FASTFLOAT_HAS_SIMD 1
+#endif
+
+#if defined(__GNUC__)
+// disable -Wcast-align=strict (GCC only)
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS \
+ _Pragma("GCC diagnostic push") \
+ _Pragma("GCC diagnostic ignored \"-Wcast-align\"")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+#endif
+
+#if defined(__GNUC__)
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS _Pragma("GCC diagnostic pop")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#define simdjson_fastfloat_really_inline __forceinline
+#else
+#define simdjson_fastfloat_really_inline inline __attribute__((always_inline))
+#endif
+
+// Branch-probability hint marking the rare slow-path branches as cold, so the
+// optimizer keeps the out-of-line slow-path re-parse off the hot path (and does
+// not duplicate the force-inlined hot scanner into the caller, which bloated
+// the hot frame and hurt ILP on some targets). Used at the call site as
+// if simdjson_fastfloat_unlikely(cond) { ... }
+// (the macro supplies the parentheses). It expands to the standard [[unlikely]]
+// attribute when supported, otherwise to __builtin_expect on GCC/Clang, or
+// to a no-op elsewhere (e.g. pre-C++20 MSVC, which has no equivalent hint).
+#ifdef __has_cpp_attribute
+#if __has_cpp_attribute(unlikely) >= 201803L
+// g++-9 hits hits this branch, but then fails to compile
+// [[unlikely]]. This happens only with g++-9.
+#if !defined(__GNUC__) || (__GNUC__ != 9)
+#define SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#endif
+#endif
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#define simdjson_fastfloat_unlikely(x) (x) [[unlikely]]
+#elif defined(__GNUC__) || defined(__clang__)
+#define simdjson_fastfloat_unlikely(x) (__builtin_expect(!!(x), 0))
+#else
+#define simdjson_fastfloat_unlikely(x) (x)
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_ASSERT
+#define SIMDJSON_FASTFLOAT_ASSERT(x) \
{ \
- std::abort(); \
+ static_cast<void>(x); \
+ }
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DEBUG_ASSERT
+#define SIMDJSON_FASTFLOAT_DEBUG_ASSERT(x) \
+ { \
+ static_cast<void>(x); \
}
+#endif
-namespace simdjson {
-namespace compile_time {
+// rust style `try!()` macro, or `?` operator
+#define SIMDJSON_FASTFLOAT_TRY(x) \
+ { \
+ if (!(x)) \
+ return false; \
+ }
-/**
- * Namespace for number parsing utilities.
- * We seek to provide exact compile-time number parsing functions.
- * That is not trivial, but thankfully we can reuse much of the existing
- * simdjson functionality.
- * Importantly, it is not a trivial matter to provide correct rounding
- * for floating-point numbers at compile-time. The fast_float library
- * does it well.
- */
-namespace number_parsing {
+#define SIMDJSON_FASTFLOAT_ENABLE_IF(...) \
+ typename std::enable_if<(__VA_ARGS__), int>::type
+
+namespace simdjson_fast_float {
+
+simdjson_fastfloat_really_inline constexpr bool cpp20_and_in_constexpr() {
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED
+ return std::is_constant_evaluated();
+#else
+ return false;
+#endif
+}
+
+template <typename T>
+struct is_supported_float_type
+ : std::integral_constant<
+ bool, std::is_same<T, double>::value || std::is_same<T, float>::value
+#ifdef __STDCPP_FLOAT64_T__
+ || std::is_same<T, std::float64_t>::value
+#endif
+#ifdef __STDCPP_FLOAT32_T__
+ || std::is_same<T, std::float32_t>::value
+#endif
+#ifdef __STDCPP_FLOAT16_T__
+ || std::is_same<T, std::float16_t>::value
+#endif
+#ifdef __STDCPP_BFLOAT16_T__
+ || std::is_same<T, std::bfloat16_t>::value
+#endif
+ > {
+};
+
+template <typename T>
+using equiv_uint_t = typename std::conditional<
+ sizeof(T) == 1, uint8_t,
+ typename std::conditional<
+ sizeof(T) == 2, uint16_t,
+ typename std::conditional<sizeof(T) == 4, uint32_t,
+ uint64_t>::type>::type>::type;
+
+template <typename T> struct is_supported_integer_type : std::is_integral<T> {};
+
+template <typename UC>
+struct is_supported_char_type
+ : std::integral_constant<bool, std::is_same<UC, char>::value ||
+ std::is_same<UC, wchar_t>::value ||
+ std::is_same<UC, char16_t>::value ||
+ std::is_same<UC, char32_t>::value
+#ifdef __cpp_char8_t
+ || std::is_same<UC, char8_t>::value
+#endif
+ > {
+};
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp3(UC const *actual_mixedcase,
+ UC const *expected_lowercase) {
+ uint64_t mask{0};
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ }
+ else {
+ return false;
+ }
+
+ uint64_t val1{0}, val2{0};
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < 3; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1 || sizeof(UC) == 2) {
+ ::memcpy(&val1, actual_mixedcase, 3 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 3 * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ return val1 == val2;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[2] | 32) == (expected_lowercase[2]);
+ }
+ else {
+ return false;
+ }
+ }
+}
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp5(UC const *actual_mixedcase,
+ UC const *expected_lowercase) {
+ uint64_t mask{0};
+ uint64_t val1{0}, val2{0};
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < 5; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) {
+ mask = 0x2020202020202020;
+ ::memcpy(&val1, actual_mixedcase, 5 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 5 * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ return val1 == val2;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ ::memcpy(&val1, actual_mixedcase, 4 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 4 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ ::memcpy(&val1, actual_mixedcase + 2, 2 * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase + 2, 2 * sizeof(UC));
+ val1 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+ }
+ else {
+ return false;
+ }
+ }
+}
+
+// Compares two ASCII strings in a case insensitive manner.
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp(UC const *actual_mixedcase, UC const *expected_lowercase,
+ size_t length) {
+ uint64_t mask{0};
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+ mask = 0x0020002000200020;
+ }
+ else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+ mask = 0x0000002000000020;
+ }
+ else {
+ return false;
+ }
+
+ if (cpp20_and_in_constexpr()) {
+ for (size_t i = 0; i < length; i++) {
+ if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+ return false;
+ }
+ }
+ return true;
+ } else {
+ uint64_t val1{0}, val2{0};
+ size_t sz{8 / (sizeof(UC))};
+ for (size_t i = 0; i < length; i += sz) {
+ val1 = val2 = 0;
+ sz = sz < (length - i) ? sz : length - i;
+ ::memcpy(&val1, actual_mixedcase + i, sz * sizeof(UC));
+ ::memcpy(&val2, expected_lowercase + i, sz * sizeof(UC));
+ val1 |= mask;
+ val2 |= mask;
+ if (val1 != val2) {
+ return false;
+ }
+ }
+ return true;
+ }
+}
+
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+
+// a pointer and a length to a contiguous block of memory
+template <typename T> struct span {
+ T const *ptr;
+ size_t length;
+
+ constexpr span(T const *_ptr, size_t _length) : ptr(_ptr), length(_length) {}
+
+ constexpr span() : ptr(nullptr), length(0) {}
+
+ constexpr size_t len() const noexcept { return length; }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const T &operator[](size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return ptr[index];
+ }
+};
+
+struct value128 {
+ uint64_t low;
+ uint64_t high;
+
+ constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
+
+ constexpr value128() : low(0), high(0) {}
+};
-// Counts the number of leading zeros in a 64-bit integer.
-consteval int leading_zeroes(uint64_t input_num, int last_bit = 0) {
+/* Helper C++14 constexpr generic implementation of leading_zeroes */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+leading_zeroes_generic(uint64_t input_num, int last_bit = 0) {
if (input_num & uint64_t(0xffffffff00000000)) {
input_num >>= 32;
last_bit |= 32;
@@ -186809,199 +240259,4540 @@ consteval int leading_zeroes(uint64_t input_num, int last_bit = 0) {
}
return 63 - last_bit;
}
-// Multiplies two 32-bit unsigned integers and returns a 64-bit result.
-consteval uint64_t emulu(uint32_t x, uint32_t y) { return x * (uint64_t)y; }
-consteval uint64_t umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
- uint64_t ad = emulu((uint32_t)(ab >> 32), (uint32_t)cd);
- uint64_t bd = emulu((uint32_t)ab, (uint32_t)cd);
- uint64_t adbc = ad + emulu((uint32_t)ab, (uint32_t)(cd >> 32));
- uint64_t adbc_carry = (uint64_t)(adbc < ad);
+
+/* result might be undefined when input_num is zero */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+leading_zeroes(uint64_t input_num) {
+ assert(input_num > 0);
+ if (cpp20_and_in_constexpr()) {
+ return leading_zeroes_generic(input_num);
+ }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#if defined(_M_X64) || defined(_M_ARM64)
+ unsigned long leading_zero = 0;
+ // Search the mask data from most significant bit (MSB)
+ // to least significant bit (LSB) for a set bit (1).
+ _BitScanReverse64(&leading_zero, input_num);
+ return static_cast<int>(63 - leading_zero);
+#else
+ return leading_zeroes_generic(input_num);
+#endif
+#else
+ return __builtin_clzll(input_num);
+#endif
+}
+
+/* Helper C++14 constexpr generic implementation of countr_zero for 32-bit */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+countr_zero_generic_32(uint32_t input_num) {
+ if (input_num == 0) {
+ return 32;
+ }
+ int last_bit = 0;
+ if (!(input_num & 0x0000FFFF)) {
+ input_num >>= 16;
+ last_bit |= 16;
+ }
+ if (!(input_num & 0x00FF)) {
+ input_num >>= 8;
+ last_bit |= 8;
+ }
+ if (!(input_num & 0x0F)) {
+ input_num >>= 4;
+ last_bit |= 4;
+ }
+ if (!(input_num & 0x3)) {
+ input_num >>= 2;
+ last_bit |= 2;
+ }
+ if (!(input_num & 0x1)) {
+ last_bit |= 1;
+ }
+ return last_bit;
+}
+
+/* count trailing zeroes for 32-bit integers */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+countr_zero_32(uint32_t input_num) {
+ if (cpp20_and_in_constexpr()) {
+ return countr_zero_generic_32(input_num);
+ }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+ unsigned long trailing_zero = 0;
+ if (_BitScanForward(&trailing_zero, input_num)) {
+ return static_cast<int>(trailing_zero);
+ }
+ return 32;
+#else
+ return input_num == 0 ? 32 : __builtin_ctz(input_num);
+#endif
+}
+
+// slow emulation routine for 32-bit
+simdjson_fastfloat_really_inline constexpr uint64_t emulu(uint32_t x, uint32_t y) {
+ return x * static_cast<uint64_t>(y);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
+ uint64_t ad =
+ emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd));
+ uint64_t bd = emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd));
+ uint64_t adbc =
+ ad + emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd >> 32));
+ uint64_t adbc_carry = static_cast<uint64_t>(adbc < ad);
uint64_t lo = bd + (adbc << 32);
- *hi = emulu((uint32_t)(ab >> 32), (uint32_t)(cd >> 32)) + (adbc >> 32) +
- (adbc_carry << 32) + (uint64_t)(lo < bd);
+ *hi =
+ emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd >> 32)) +
+ (adbc >> 32) + (adbc_carry << 32) + static_cast<uint64_t>(lo < bd);
return lo;
}
-// Represents a 128-bit unsigned integer as two 64-bit parts.
-// We have a value128 struct elsewhere in the simdjson, but we
-// use a separate one here for clarity.
-struct value128 {
- uint64_t low;
- uint64_t high;
+#ifdef SIMDJSON_FASTFLOAT_32BIT
- constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
- constexpr value128() : low(0), high(0) {}
-};
+// slow emulation routine for 32-bit
+#if !defined(__MINGW64__)
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t _umul128(uint64_t ab,
+ uint64_t cd,
+ uint64_t *hi) {
+ return umul128_generic(ab, cd, hi);
+}
+#endif // !__MINGW64__
+
+#endif // SIMDJSON_FASTFLOAT_32BIT
-// Multiplies two 64-bit integers and returns a 128-bit result as value128.
-consteval value128 full_multiplication(uint64_t a, uint64_t b) {
+// compute 64-bit a*b
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+full_multiplication(uint64_t a, uint64_t b) {
+ if (cpp20_and_in_constexpr()) {
+ value128 answer;
+ answer.low = umul128_generic(a, b, &answer.high);
+ return answer;
+ }
value128 answer;
+#if defined(_M_ARM64) && !defined(__MINGW32__)
+ // ARM64 has native support for 64-bit multiplications, no need to emulate
+ // But MinGW on ARM64 doesn't have native support for 64-bit multiplications
+ answer.high = __umulh(a, b);
+ answer.low = a * b;
+#elif defined(SIMDJSON_FASTFLOAT_32BIT) || (defined(_WIN64) && !defined(__clang__) && \
+ !defined(_M_ARM64) && !defined(__GNUC__))
+ answer.low = _umul128(a, b, &answer.high); // _umul128 not available on ARM64
+#elif defined(SIMDJSON_FASTFLOAT_64BIT) && defined(__SIZEOF_INT128__)
+ __uint128_t r = static_cast<__uint128_t>(a) * b;
+ answer.low = uint64_t(r);
+ answer.high = uint64_t(r >> 64);
+#else
answer.low = umul128_generic(a, b, &answer.high);
+#endif
return answer;
}
-// Converts mantissa and exponent to a double, considering the sign.
-consteval double to_double(uint64_t mantissa, int64_t exponent, bool negative) {
- uint64_t sign_bit = negative ? (1ULL << 63) : 0;
- uint64_t exponent_bits = (uint64_t(exponent) & 0x7FF) << 52;
- uint64_t bits = sign_bit | exponent_bits | (mantissa & ((1ULL << 52) - 1));
- return std::bit_cast<double>(bits);
+struct adjusted_mantissa {
+ uint64_t mantissa{0};
+ int32_t power2{0}; // a negative value indicates an invalid result
+ adjusted_mantissa() = default;
+
+ constexpr bool operator==(adjusted_mantissa const &o) const {
+ return mantissa == o.mantissa && power2 == o.power2;
+ }
+
+ constexpr bool operator!=(adjusted_mantissa const &o) const {
+ return mantissa != o.mantissa || power2 != o.power2;
+ }
+};
+
+// Bias so we can get the real exponent with an invalid adjusted_mantissa.
+constexpr static int32_t invalid_am_bias = -0x8000;
+
+// used for binary_format_lookup_tables<T>::max_mantissa
+constexpr uint64_t constant_55555 = 5 * 5 * 5 * 5 * 5;
+
+template <typename T, typename U = void> struct binary_format_lookup_tables;
+
+template <typename T> struct binary_format : binary_format_lookup_tables<T> {
+ using equiv_uint = equiv_uint_t<T>;
+
+ static constexpr int mantissa_explicit_bits();
+ static constexpr int minimum_exponent();
+ static constexpr int infinite_power();
+ static constexpr int sign_index();
+ static constexpr int
+ min_exponent_fast_path(); // used when fegetround() == FE_TONEAREST
+ static constexpr int max_exponent_fast_path();
+ static constexpr int max_exponent_round_to_even();
+ static constexpr int min_exponent_round_to_even();
+ static constexpr uint64_t max_mantissa_fast_path(int64_t power);
+ static constexpr uint64_t
+ max_mantissa_fast_path(); // used when fegetround() == FE_TONEAREST
+ static constexpr int largest_power_of_ten();
+ static constexpr int smallest_power_of_ten();
+ static constexpr T exact_power_of_ten(int64_t power);
+ static constexpr size_t max_digits();
+ static constexpr equiv_uint exponent_mask();
+ static constexpr equiv_uint mantissa_mask();
+ static constexpr equiv_uint hidden_bit_mask();
+};
+
+template <typename U> struct binary_format_lookup_tables<double, U> {
+ static constexpr double powers_of_ten[] = {
+ 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
+ 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22};
+
+ // Largest integer value v so that (5**index * v) <= 1<<53.
+ // 0x20000000000000 == 1 << 53
+ static constexpr uint64_t max_mantissa[] = {
+ 0x20000000000000,
+ 0x20000000000000 / 5,
+ 0x20000000000000 / (5 * 5),
+ 0x20000000000000 / (5 * 5 * 5),
+ 0x20000000000000 / (5 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555),
+ 0x20000000000000 / (constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * 5 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5 * 5),
+ 0x20000000000000 /
+ (constant_55555 * constant_55555 * constant_55555 * constant_55555),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5 * 5),
+ 0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+ constant_55555 * 5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr double binary_format_lookup_tables<double, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<double, U>::max_mantissa[];
+
+#endif
+
+template <typename U> struct binary_format_lookup_tables<float, U> {
+ static constexpr float powers_of_ten[] = {1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+ 1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+
+ // Largest integer value v so that (5**index * v) <= 1<<24.
+ // 0x1000000 == 1<<24
+ static constexpr uint64_t max_mantissa[] = {
+ 0x1000000,
+ 0x1000000 / 5,
+ 0x1000000 / (5 * 5),
+ 0x1000000 / (5 * 5 * 5),
+ 0x1000000 / (5 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555),
+ 0x1000000 / (constant_55555 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555 * 5 * 5 * 5 * 5),
+ 0x1000000 / (constant_55555 * constant_55555),
+ 0x1000000 / (constant_55555 * constant_55555 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr float binary_format_lookup_tables<float, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<float, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return 0;
+#else
+ return -22;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return 0;
+#else
+ return -10;
+#endif
}
-// Attempts to compute i * 10^(power) exactly; and if "negative" is
-// true, negate the result.
-// Returns true on success, false on failure.
-// Failure suggests and invalid input or out-of-range result.
-consteval bool compute_float_64(int64_t power, uint64_t i, bool negative,
- double &d) {
- if (i == 0) {
- d = negative ? -0.0 : 0.0;
+template <>
+inline constexpr int binary_format<double>::mantissa_explicit_bits() {
+ return 52;
+}
+
+template <>
+inline constexpr int binary_format<float>::mantissa_explicit_bits() {
+ return 23;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_round_to_even() {
+ return 23;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_round_to_even() {
+ return 10;
+}
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_round_to_even() {
+ return -4;
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_round_to_even() {
+ return -17;
+}
+
+template <> inline constexpr int binary_format<double>::minimum_exponent() {
+ return -1023;
+}
+
+template <> inline constexpr int binary_format<float>::minimum_exponent() {
+ return -127;
+}
+
+template <> inline constexpr int binary_format<double>::infinite_power() {
+ return 0x7FF;
+}
+
+template <> inline constexpr int binary_format<float>::infinite_power() {
+ return 0xFF;
+}
+
+template <> inline constexpr int binary_format<double>::sign_index() {
+ return 63;
+}
+
+template <> inline constexpr int binary_format<float>::sign_index() {
+ return 31;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_fast_path() {
+ return 22;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_fast_path() {
+ return 10;
+}
+
+template <>
+inline constexpr uint64_t binary_format<double>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t binary_format<float>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_FLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::float16_t, U> {
+ static constexpr std::float16_t powers_of_ten[] = {1e0f16, 1e1f16, 1e2f16,
+ 1e3f16, 1e4f16};
+
+ // Largest integer value v so that (5**index * v) <= 1<<11.
+ // 0x800 == 1<<11
+ static constexpr uint64_t max_mantissa[] = {0x800,
+ 0x800 / 5,
+ 0x800 / (5 * 5),
+ 0x800 / (5 * 5 * 5),
+ 0x800 / (5 * 5 * 5 * 5),
+ 0x800 / (constant_55555)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::float16_t
+ binary_format_lookup_tables<std::float16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+ binary_format_lookup_tables<std::float16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::float16_t
+binary_format<std::float16_t>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::exponent_mask() {
+ return 0x7C00;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::mantissa_mask() {
+ return 0x03FF;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::hidden_bit_mask() {
+ return 0x0400;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::max_exponent_fast_path() {
+ return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::mantissa_explicit_bits() {
+ return 10;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 4
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::min_exponent_fast_path() {
+ return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::max_exponent_round_to_even() {
+ return 5;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::min_exponent_round_to_even() {
+ return -22;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::minimum_exponent() {
+ return -15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::infinite_power() {
+ return 0x1F;
+}
+
+template <> inline constexpr int binary_format<std::float16_t>::sign_index() {
+ return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::largest_power_of_ten() {
+ return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::smallest_power_of_ten() {
+ return -27;
+}
+
+template <>
+inline constexpr size_t binary_format<std::float16_t>::max_digits() {
+ return 22;
+}
+#endif // __STDCPP_FLOAT16_T__
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_BFLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::bfloat16_t, U> {
+ static constexpr std::bfloat16_t powers_of_ten[] = {1e0bf16, 1e1bf16, 1e2bf16,
+ 1e3bf16};
+
+ // Largest integer value v so that (5**index * v) <= 1<<8.
+ // 0x100 == 1<<8
+ static constexpr uint64_t max_mantissa[] = {0x100, 0x100 / 5, 0x100 / (5 * 5),
+ 0x100 / (5 * 5 * 5),
+ 0x100 / (5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::bfloat16_t
+ binary_format_lookup_tables<std::bfloat16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+ binary_format_lookup_tables<std::bfloat16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::bfloat16_t
+binary_format<std::bfloat16_t>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::max_exponent_fast_path() {
+ return 3;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::exponent_mask() {
+ return 0x7F80;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::mantissa_mask() {
+ return 0x007F;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::hidden_bit_mask() {
+ return 0x0080;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::mantissa_explicit_bits() {
+ return 7;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path() {
+ return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 3
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::min_exponent_fast_path() {
+ return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::max_exponent_round_to_even() {
+ return 3;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::min_exponent_round_to_even() {
+ return -24;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::minimum_exponent() {
+ return -127;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::infinite_power() {
+ return 0xFF;
+}
+
+template <> inline constexpr int binary_format<std::bfloat16_t>::sign_index() {
+ return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::largest_power_of_ten() {
+ return 38;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::smallest_power_of_ten() {
+ return -60;
+}
+
+template <>
+inline constexpr size_t binary_format<std::bfloat16_t>::max_digits() {
+ return 98;
+}
+#endif // __STDCPP_BFLOAT16_T__
+
+template <>
+inline constexpr uint64_t
+binary_format<double>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 22
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<float>::max_mantissa_fast_path(int64_t power) {
+ // caller is responsible to ensure that
+ // power >= 0 && power <= 10
+ //
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr double
+binary_format<double>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr float binary_format<float>::exact_power_of_ten(int64_t power) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <> inline constexpr int binary_format<double>::largest_power_of_ten() {
+ return 308;
+}
+
+template <> inline constexpr int binary_format<float>::largest_power_of_ten() {
+ return 38;
+}
+
+template <>
+inline constexpr int binary_format<double>::smallest_power_of_ten() {
+ return -342;
+}
+
+template <> inline constexpr int binary_format<float>::smallest_power_of_ten() {
+ return -64;
+}
+
+template <> inline constexpr size_t binary_format<double>::max_digits() {
+ return 769;
+}
+
+template <> inline constexpr size_t binary_format<float>::max_digits() {
+ return 114;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::exponent_mask() {
+ return 0x7F800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::exponent_mask() {
+ return 0x7FF0000000000000;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::mantissa_mask() {
+ return 0x007FFFFF;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::mantissa_mask() {
+ return 0x000FFFFFFFFFFFFF;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::hidden_bit_mask() {
+ return 0x00800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::hidden_bit_mask() {
+ return 0x0010000000000000;
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+to_float(bool negative, adjusted_mantissa am, T &value) {
+ using equiv_uint = equiv_uint_t<T>;
+ equiv_uint word = equiv_uint(am.mantissa);
+ word = equiv_uint(word | equiv_uint(am.power2)
+ << binary_format<T>::mantissa_explicit_bits());
+ word =
+ equiv_uint(word | equiv_uint(negative) << binary_format<T>::sign_index());
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ value = std::bit_cast<T>(word);
+#else
+ ::memcpy(&value, &word, sizeof(T));
+#endif
+}
+
+template <typename = void> struct space_lut {
+ static constexpr bool value[] = {
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr bool space_lut<T>::value[];
+
+#endif
+
+template <typename UC> constexpr bool is_space(UC c) {
+ // wchar_t and char can be signed, so a negative code unit slips past a plain
+ // `c < 256` and then indexes the table by its truncated low byte. Compare as
+ // unsigned, matching the care taken in ch_to_digit.
+ using UnsignedUC = typename std::make_unsigned<UC>::type;
+ return static_cast<UnsignedUC>(c) < 256 && space_lut<>::value[uint8_t(c)];
+}
+
+template <typename UC> static constexpr uint64_t int_cmp_zeros() {
+ static_assert((sizeof(UC) == 1) || (sizeof(UC) == 2) || (sizeof(UC) == 4),
+ "Unsupported character size");
+ return (sizeof(UC) == 1) ? 0x3030303030303030
+ : (sizeof(UC) == 2)
+ ? (uint64_t(UC('0')) << 48 | uint64_t(UC('0')) << 32 |
+ uint64_t(UC('0')) << 16 | UC('0'))
+ : (uint64_t(UC('0')) << 32 | UC('0'));
+}
+
+template <typename UC> static constexpr int int_cmp_len() {
+ return sizeof(uint64_t) / sizeof(UC);
+}
+
+template <typename UC> constexpr UC const *str_const_nan();
+
+template <> constexpr char const *str_const_nan<char>() { return "nan"; }
+
+template <> constexpr wchar_t const *str_const_nan<wchar_t>() { return L"nan"; }
+
+template <> constexpr char16_t const *str_const_nan<char16_t>() {
+ return u"nan";
+}
+
+template <> constexpr char32_t const *str_const_nan<char32_t>() {
+ return U"nan";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_nan<char8_t>() {
+ return u8"nan";
+}
+#endif
+
+template <typename UC> constexpr UC const *str_const_inf();
+
+template <> constexpr char const *str_const_inf<char>() { return "infinity"; }
+
+template <> constexpr wchar_t const *str_const_inf<wchar_t>() {
+ return L"infinity";
+}
+
+template <> constexpr char16_t const *str_const_inf<char16_t>() {
+ return u"infinity";
+}
+
+template <> constexpr char32_t const *str_const_inf<char32_t>() {
+ return U"infinity";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_inf<char8_t>() {
+ return u8"infinity";
+}
+#endif
+
+template <typename = void> struct int_luts {
+ static constexpr uint8_t chdigit[] = {
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 255, 255,
+ 255, 255, 255, 255, 255, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19,
+ 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34,
+ 35, 255, 255, 255, 255, 255, 255, 10, 11, 12, 13, 14, 15, 16, 17,
+ 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32,
+ 33, 34, 35, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+ 255};
+
+ static constexpr size_t maxdigits_u64[] = {
+ 64, 41, 32, 28, 25, 23, 22, 21, 20, 19, 18, 18, 17, 17, 16, 16, 16, 16,
+ 15, 15, 15, 15, 14, 14, 14, 14, 14, 14, 14, 13, 13, 13, 13, 13, 13};
+
+ static constexpr uint64_t min_safe_u64[] = {
+ 9223372036854775808ull, 12157665459056928801ull, 4611686018427387904,
+ 7450580596923828125, 4738381338321616896, 3909821048582988049,
+ 9223372036854775808ull, 12157665459056928801ull, 10000000000000000000ull,
+ 5559917313492231481, 2218611106740436992, 8650415919381337933,
+ 2177953337809371136, 6568408355712890625, 1152921504606846976,
+ 2862423051509815793, 6746640616477458432, 15181127029874798299ull,
+ 1638400000000000000, 3243919932521508681, 6221821273427820544,
+ 11592836324538749809ull, 876488338465357824, 1490116119384765625,
+ 2481152873203736576, 4052555153018976267, 6502111422497947648,
+ 10260628712958602189ull, 15943230000000000000ull, 787662783788549761,
+ 1152921504606846976, 1667889514952984961, 2386420683693101056,
+ 3379220508056640625, 4738381338321616896};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint8_t int_luts<T>::chdigit[];
+
+template <typename T> constexpr size_t int_luts<T>::maxdigits_u64[];
+
+template <typename T> constexpr uint64_t int_luts<T>::min_safe_u64[];
+
+#endif
+
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr uint8_t ch_to_digit(UC c) {
+ // wchar_t and char can be signed, so we need to be careful.
+ using UnsignedUC = typename std::make_unsigned<UC>::type;
+ return int_luts<>::chdigit[static_cast<unsigned char>(
+ static_cast<UnsignedUC>(c) &
+ static_cast<UnsignedUC>(
+ -((static_cast<UnsignedUC>(c) & ~0xFFull) == 0)))];
+}
+
+simdjson_fastfloat_really_inline constexpr size_t max_digits_u64(int base) {
+ return int_luts<>::maxdigits_u64[base - 2];
+}
+
+// If a u64 is exactly max_digits_u64() in length, this is
+// the value below which it has definitely overflowed.
+simdjson_fastfloat_really_inline constexpr uint64_t min_safe_u64(int base) {
+ return int_luts<>::min_safe_u64[base - 2];
+}
+
+static_assert(std::is_same<equiv_uint_t<double>, uint64_t>::value,
+ "equiv_uint should be uint64_t for double");
+static_assert(std::numeric_limits<double>::is_iec559,
+ "double must fulfill the requirements of IEC 559 (IEEE 754)");
+
+static_assert(std::is_same<equiv_uint_t<float>, uint32_t>::value,
+ "equiv_uint should be uint32_t for float");
+static_assert(std::numeric_limits<float>::is_iec559,
+ "float must fulfill the requirements of IEC 559 (IEEE 754)");
+
+#ifdef __STDCPP_FLOAT64_T__
+static_assert(std::is_same<equiv_uint_t<std::float64_t>, uint64_t>::value,
+ "equiv_uint should be uint64_t for std::float64_t");
+static_assert(
+ std::numeric_limits<std::float64_t>::is_iec559,
+ "std::float64_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float64_t> : public binary_format<double> {};
+#endif // __STDCPP_FLOAT64_T__
+
+#ifdef __STDCPP_FLOAT32_T__
+static_assert(std::is_same<equiv_uint_t<std::float32_t>, uint32_t>::value,
+ "equiv_uint should be uint32_t for std::float32_t");
+static_assert(
+ std::numeric_limits<std::float32_t>::is_iec559,
+ "std::float32_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float32_t> : public binary_format<float> {};
+#endif // __STDCPP_FLOAT32_T__
+
+#ifdef __STDCPP_FLOAT16_T__
+static_assert(
+ std::is_same<binary_format<std::float16_t>::equiv_uint, uint16_t>::value,
+ "equiv_uint should be uint16_t for std::float16_t");
+static_assert(
+ std::numeric_limits<std::float16_t>::is_iec559,
+ "std::float16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_FLOAT16_T__
+
+#ifdef __STDCPP_BFLOAT16_T__
+static_assert(
+ std::is_same<binary_format<std::bfloat16_t>::equiv_uint, uint16_t>::value,
+ "equiv_uint should be uint16_t for std::bfloat16_t");
+static_assert(
+ std::numeric_limits<std::bfloat16_t>::is_iec559,
+ "std::bfloat16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_BFLOAT16_T__
+
+constexpr chars_format operator~(chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(~static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator&(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) &
+ static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator|(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) |
+ static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator^(chars_format lhs, chars_format rhs) noexcept {
+ using int_type = std::underlying_type<chars_format>::type;
+ return static_cast<chars_format>(static_cast<int_type>(lhs) ^
+ static_cast<int_type>(rhs));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator&=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs & rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator|=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs | rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator^=(chars_format &lhs, chars_format rhs) noexcept {
+ return lhs = (lhs ^ rhs);
+}
+
+namespace detail {
+// adjust for deprecated feature macros
+constexpr chars_format adjust_for_feature_macros(chars_format fmt) {
+ return fmt
+#ifdef SIMDJSON_FASTFLOAT_ALLOWS_LEADING_PLUS
+ | chars_format::allow_leading_plus
+#endif
+#ifdef SIMDJSON_FASTFLOAT_SKIP_WHITE_SPACE
+ | chars_format::skip_white_space
+#endif
+ ;
+}
+} // namespace detail
+} // namespace simdjson_fast_float
+
+#endif
+
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+#define SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+
+namespace simdjson_fast_float {
+/**
+ * This function parses the character sequence [first,last) for a number. It
+ * parses floating-point numbers expecting a locale-independent format
+ * equivalent to what is used by std::strtod in the default ("C") locale. The
+ * resulting floating-point value is the closest floating-point values (using
+ * either float or double), using the "round to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * parsing according to the IEEE standard.
+ *
+ * Given a successful parse, the pointer (`ptr`) in the returned value is set to
+ * point right after the parsed number, and the `value` referenced is set to the
+ * parsed value. In case of error, the returned `ec` contains a representative
+ * error, otherwise the default (`std::errc()`) value is stored.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ *
+ * Like the C++17 standard, the `simdjson_fast_float::from_chars` functions take an
+ * optional last argument of the type `simdjson_fast_float::chars_format`. It is a bitset
+ * value: we check whether `fmt & simdjson_fast_float::chars_format::fixed` and `fmt &
+ * simdjson_fast_float::chars_format::scientific` are set to determine whether we allow
+ * the fixed point and scientific notation respectively. The default is
+ * `simdjson_fast_float::chars_format::general` which allows both `fixed` and
+ * `scientific`.
+ */
+template <typename T, typename UC = char,
+ typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_float_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+ chars_format fmt = chars_format::general) noexcept;
+
+/**
+ * Like from_chars, but accepts an `options` argument to govern number parsing.
+ * Both for floating-point types and integer types.
+ */
+template <typename T, typename UC = char>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept;
+
+/**
+ * This function multiplies an integer number by a power of 10 and returns
+ * the result as a double precision floating-point value that is correctly
+ * rounded. The resulting floating-point value is the closest floating-point
+ * value, using the "round to nearest, tie to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * conversion according to the IEEE standard.
+ *
+ * On overflow infinity is returned, on underflow 0 is returned.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ */
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * This function is a template overload of `integer_times_pow10()`
+ * that returns a floating-point value of type `T` that is one of
+ * supported floating-point types (e.g. `double`, `float`).
+ */
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * from_chars for integer types.
+ */
+template <typename T, typename UC = char,
+ typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_integer_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base = 10) noexcept;
+
+} // namespace simdjson_fast_float
+
+#endif // SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+#ifndef SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+#define SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+
+#include <cctype>
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+#include <limits>
+#include <type_traits>
+
+
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+#include <emmintrin.h>
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_NEON
+#include <arm_neon.h>
+#endif
+
+namespace simdjson_fast_float {
+
+template <typename UC> simdjson_fastfloat_really_inline constexpr bool has_simd_opt() {
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+ return std::is_same<UC, char16_t>::value;
+#else
+ return false;
+#endif
+}
+
+// Next function can be micro-optimized, but compilers are entirely
+// able to optimize it well.
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr bool is_integer(UC c) noexcept {
+ return static_cast<unsigned>(c - UC('0')) <= 9u;
+}
+
+simdjson_fastfloat_really_inline constexpr uint64_t byteswap(uint64_t val) {
+ return (val & 0xFF00000000000000) >> 56 | (val & 0x00FF000000000000) >> 40 |
+ (val & 0x0000FF0000000000) >> 24 | (val & 0x000000FF00000000) >> 8 |
+ (val & 0x00000000FF000000) << 8 | (val & 0x0000000000FF0000) << 24 |
+ (val & 0x000000000000FF00) << 40 | (val & 0x00000000000000FF) << 56;
+}
+
+simdjson_fastfloat_really_inline constexpr uint32_t byteswap_32(uint32_t val) {
+ return (val >> 24) | ((val >> 8) & 0x0000FF00u) | ((val << 8) & 0x00FF0000u) |
+ (val << 24);
+}
+
+// Read 8 UC into a u64. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+read8_to_u64(UC const *chars) {
+ if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+ uint64_t val = 0;
+ for (int i = 0; i < 8; ++i) {
+ val |= uint64_t(uint8_t(*chars)) << (i * 8);
+ ++chars;
+ }
+ return val;
+ }
+ uint64_t val;
+ ::memcpy(&val, chars, sizeof(uint64_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+ // Need to read as-if the number was in little-endian order.
+ val = byteswap(val);
+#endif
+ return val;
+}
+
+// Read 4 UC into a u32. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+read4_to_u32(UC const *chars) {
+ if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+ uint32_t val = 0;
+ for (int i = 0; i < 4; ++i) {
+ val |= uint32_t(uint8_t(*chars)) << (i * 8);
+ ++chars;
+ }
+ return val;
+ }
+ uint32_t val;
+ ::memcpy(&val, chars, sizeof(uint32_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+ val = byteswap_32(val);
+#endif
+ return val;
+}
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(__m128i const data) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ __m128i const packed = _mm_packus_epi16(data, data);
+#ifdef SIMDJSON_FASTFLOAT_64BIT
+ return uint64_t(_mm_cvtsi128_si64(packed));
+#else
+ uint64_t value;
+ // Visual Studio + older versions of GCC don't support _mm_storeu_si64
+ _mm_storel_epi64(reinterpret_cast<__m128i *>(&value), packed);
+ return value;
+#endif
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ return simd_read8_to_u64(
+ _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars)));
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(uint16x8_t const data) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ uint8x8_t utf8_packed = vmovn_u16(data);
+ return vget_lane_u64(vreinterpret_u64_u8(utf8_packed), 0);
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ return simd_read8_to_u64(
+ vld1q_u16(reinterpret_cast<uint16_t const *>(chars)));
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#endif // SIMDJSON_FASTFLOAT_SSE2
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+uint64_t simd_read8_to_u64(UC const *) {
+ return 0;
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_eight_digits_unrolled(uint64_t val) {
+ uint64_t const mask = 0x000000FF000000FF;
+ uint64_t const mul1 = 0x000F424000000064; // 100 + (1000000ULL << 32)
+ uint64_t const mul2 = 0x0000271000000001; // 1 + (10000ULL << 32)
+ val -= 0x3030303030303030;
+ val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
+ val = (((val & mask) * mul1) + (((val >> 16) & mask) * mul2)) >> 32;
+ return uint32_t(val);
+}
+
+// Call this if chars are definitely 8 digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+parse_eight_digits_unrolled(UC const *chars) noexcept {
+ if (cpp20_and_in_constexpr() || !has_simd_opt<UC>()) {
+ return parse_eight_digits_unrolled(read8_to_u64(chars)); // truncation okay
+ }
+ return parse_eight_digits_unrolled(simd_read8_to_u64(chars));
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_eight_digits_fast(uint64_t val) noexcept {
+ return !((((val + 0x4646464646464646) | (val - 0x3030303030303030)) &
+ 0x8080808080808080));
+}
+
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_four_digits_fast(uint32_t val) noexcept {
+ return !((((val + 0x46464646) | (val - 0x30303030)) & 0x80808080));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_four_digits_unrolled(uint32_t val) noexcept {
+ val -= 0x30303030;
+ val = (val * 10) + (val >> 8);
+ return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// Call this if chars might not be 8 digits.
+// Using this style (instead of is_made_of_eight_digits_fast() then
+// parse_eight_digits_unrolled()) ensures we don't load SIMD registers twice.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+simd_parse_if_eight_digits_unrolled(char16_t const *chars,
+ uint64_t &i) noexcept {
+ if (cpp20_and_in_constexpr()) {
+ return false;
+ }
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ __m128i const data =
+ _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars));
+
+ // (x - '0') <= 9
+ // http://0x80.pl/articles/simd-parsing-int-sequences.html
+ __m128i const t0 = _mm_add_epi16(data, _mm_set1_epi16(32720));
+ __m128i const t1 = _mm_cmpgt_epi16(t0, _mm_set1_epi16(-32759));
+
+ if (_mm_movemask_epi8(t1) == 0) {
+ i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
return true;
+ } else
+ return false;
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+ SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+ uint16x8_t const data = vld1q_u16(reinterpret_cast<uint16_t const *>(chars));
+
+ // (x - '0') <= 9
+ // http://0x80.pl/articles/simd-parsing-int-sequences.html
+ uint16x8_t const t0 = vsubq_u16(data, vmovq_n_u16('0'));
+ uint16x8_t const mask = vcltq_u16(t0, vmovq_n_u16('9' - '0' + 1));
+
+ if (vminvq_u16(mask) == 0xFFFF) {
+ i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+ return true;
+ } else
+ return false;
+ SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#else
+ static_cast<void>(chars);
+ static_cast<void>(i);
+ return false;
+#endif // SIMDJSON_FASTFLOAT_SSE2
+}
+
+#endif // SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+bool simd_parse_if_eight_digits_unrolled(UC const *, uint64_t &) {
+ return 0;
+}
+
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!std::is_same<UC, char>::value) = 0>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(UC const *&p, UC const *const pend, uint64_t &i) {
+ if (!has_simd_opt<UC>()) {
+ return;
}
- int64_t exponent = (((152170 + 65536) * power) >> 16) + 1024 + 63;
- int lz = leading_zeroes(i);
- i <<= lz;
- const uint32_t index =
- 2 * uint32_t(power - simdjson::internal::smallest_power);
- value128 firstproduct = full_multiplication(
- i, simdjson::internal::powers_template<>::power_of_five_128[index]);
- if ((firstproduct.high & 0x1FF) == 0x1FF) {
- value128 secondproduct = full_multiplication(
- i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+ while ((std::distance(p, pend) >= 8) &&
+ simd_parse_if_eight_digits_unrolled(
+ p, i)) { // in rare cases, this will overflow, but that's ok
+ p += 8;
+ }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(char const *&p, char const *const pend,
+ uint64_t &i) {
+ // optimizes better than parse_if_eight_digits_unrolled() for UC = char.
+ while ((std::distance(p, pend) >= 8) &&
+ is_made_of_eight_digits_fast(read8_to_u64(p))) {
+ i = i * 100000000 +
+ parse_eight_digits_unrolled(read8_to_u64(
+ p)); // in rare cases, this will overflow, but that's ok
+ p += 8;
+ }
+ // Consume a remaining 4-7 digit run in a single SWAR step instead of
+ // byte-by-byte (reuses the existing 4-digit helpers). The parsed result is
+ // identical either way. Historically gated to clang because gcc regressed on
+ // short remainders, but that verdict predates the span-elision restructure;
+ // with the leaner hot path the 4-digit step now wins on gcc as well.
+ if ((pend - p) >= 4) {
+ uint32_t const val4 = read4_to_u32(p);
+ if (is_made_of_four_digits_fast(val4)) {
+ i = i * 10000 +
+ parse_four_digits_unrolled(val4); // may overflow, that's ok
+ p += 4;
+ }
+ }
+}
+
+enum class parse_error {
+ no_error,
+ // [JSON-only] The minus sign must be followed by an integer.
+ missing_integer_after_sign,
+ // A sign must be followed by an integer or dot.
+ missing_integer_or_dot_after_sign,
+ // [JSON-only] The integer part must not have leading zeros.
+ leading_zeros_in_integer_part,
+ // [JSON-only] The integer part must have at least one digit.
+ no_digits_in_integer_part,
+ // [JSON-only] If there is a decimal point, there must be digits in the
+ // fractional part.
+ no_digits_in_fractional_part,
+ // The mantissa must have at least one digit.
+ no_digits_in_mantissa,
+ // Scientific notation requires an exponential part.
+ missing_exponential_part,
+};
+
+template <typename UC> struct parsed_number_string_t {
+ int64_t exponent{0};
+ uint64_t mantissa{0};
+ UC const *lastmatch{nullptr};
+ bool negative{false};
+ bool valid{false};
+ bool too_many_digits{false};
+ // contains the range of the significant digits
+ span<UC const> integer{}; // non-nullable
+ span<UC const> fraction{}; // nullable
+ parse_error error{parse_error::no_error};
+};
+
+using byte_span = span<char const>;
+using parsed_number_string = parsed_number_string_t<char>;
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+report_parse_error(UC const *p, parse_error error) {
+ parsed_number_string_t<UC> answer;
+ answer.valid = false;
+ answer.lastmatch = p;
+ answer.error = error;
+ return answer;
+}
+
+// Assuming that you use no more than 19 digits, this will
+// parse an ASCII string.
+//
+// store_spans is a *runtime* flag (not a template parameter, deliberately: a
+// template would create a second instantiation of this whole function and the
+// extra icache pressure wipes out the gain). When false, the integer/fraction
+// spans (read only by the rare digit_comp slow path) are not materialized,
+// which keeps the fat parsed_number_string_t off the hot path. The caller
+// re-parses with store_spans=true if the slow path is actually reached.
+template <bool basic_json_fmt, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+parse_number_string(UC const *p, UC const *pend, parse_options_t<UC> options,
+ bool store_spans = true) noexcept {
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ UC const decimal_point = options.decimal_point;
+
+ parsed_number_string_t<UC> answer;
+ answer.valid = false;
+ answer.too_many_digits = false;
+ // assume p < pend, so dereference without checks;
+ answer.negative = (*p == UC('-'));
+ // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+ if ((*p == UC('-')) || (uint64_t(fmt & chars_format::allow_leading_plus) &&
+ !basic_json_fmt && *p == UC('+'))) {
+ ++p;
+ if (p == pend) {
+ return report_parse_error<UC>(
+ p, parse_error::missing_integer_or_dot_after_sign);
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ if (!is_integer(*p)) { // a sign must be followed by an integer
+ return report_parse_error<UC>(p,
+ parse_error::missing_integer_after_sign);
+ }
+ }
+ else {
+ if (!is_integer(*p) &&
+ (*p !=
+ decimal_point)) { // a sign must be followed by an integer or the dot
+ return report_parse_error<UC>(
+ p, parse_error::missing_integer_or_dot_after_sign);
+ }
+ }
+ }
+ UC const *const start_digits = p;
+
+ uint64_t i = 0; // an unsigned int avoids signed overflows (which are bad)
+
+ // Straight-line unroll of the integer-part scan: most integer parts are
+ // 1-5 digits, so peeling the first iterations eliminates the loop back-edge
+ // for the common case. Semantics are identical to the original `while` loop:
+ // i = 10*i + digit, advancing p.
+ if ((p != pend) && is_integer(*p)) {
+ i = uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ if ((p != pend) && is_integer(*p)) {
+ i = 10 * i + uint64_t(*p - UC('0'));
+ ++p;
+ while ((p != pend) && is_integer(*p)) {
+ // a multiplication by 10 is cheaper than an arbitrary integer
+ // multiplication
+ i = 10 * i +
+ uint64_t(*p - UC('0')); // might overflow, handled later
+ ++p;
+ }
+ }
+ }
+ }
+ }
+ }
+ UC const *const end_of_integer_part = p;
+ int64_t digit_count = int64_t(end_of_integer_part - start_digits);
+ if (store_spans) {
+ answer.integer = span<UC const>(start_digits, size_t(digit_count));
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ // at least 1 digit in integer part, without leading zeros
+ if (digit_count == 0) {
+ return report_parse_error<UC>(p, parse_error::no_digits_in_integer_part);
+ }
+ if ((start_digits[0] == UC('0') && digit_count > 1)) {
+ return report_parse_error<UC>(start_digits,
+ parse_error::leading_zeros_in_integer_part);
+ }
+ }
+
+ int64_t exponent = 0;
+ bool const has_decimal_point = (p != pend) && (*p == decimal_point);
+ if (has_decimal_point) {
+ ++p;
+ UC const *before = p;
+ // can occur at most twice without overflowing, but let it occur more, since
+ // for integers with many digits, digit parsing is the primary bottleneck.
+ loop_parse_if_eight_digits(p, pend, i);
+
+ while ((p != pend) && is_integer(*p)) {
+ uint8_t digit = uint8_t(*p - UC('0'));
+ ++p;
+ i = i * 10 + digit; // in rare cases, this will overflow, but that's ok
+ }
+ exponent = before - p;
+ if (store_spans) {
+ answer.fraction = span<UC const>(before, size_t(p - before));
+ }
+ digit_count -= exponent;
+ }
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+ // at least 1 digit in fractional part
+ if (has_decimal_point && exponent == 0) {
+ return report_parse_error<UC>(p,
+ parse_error::no_digits_in_fractional_part);
+ }
+ }
+ else if (digit_count == 0) { // we must have encountered at least one integer!
+ return report_parse_error<UC>(p, parse_error::no_digits_in_mantissa);
+ }
+ int64_t exp_number = 0; // explicit exponential part
+ if ((uint64_t(fmt & chars_format::scientific) && (p != pend) &&
+ ((UC('e') == *p) || (UC('E') == *p))) ||
+ (uint64_t(fmt & detail::basic_fortran_fmt) && (p != pend) &&
+ ((UC('+') == *p) || (UC('-') == *p) || (UC('d') == *p) ||
+ (UC('D') == *p)))) {
+ UC const *location_of_e = p;
+ if ((UC('e') == *p) || (UC('E') == *p) || (UC('d') == *p) ||
+ (UC('D') == *p)) {
+ ++p;
+ }
+ bool neg_exp = false;
+ if ((p != pend) && (UC('-') == *p)) {
+ neg_exp = true;
+ ++p;
+ } else if ((p != pend) &&
+ (UC('+') ==
+ *p)) { // '+' on exponent is allowed by C++17 20.19.3.(7.1)
+ ++p;
+ }
+ if ((p == pend) || !is_integer(*p)) {
+ if (!uint64_t(fmt & chars_format::fixed)) {
+ // The exponential part is invalid for scientific notation, so it must
+ // be a trailing token for fixed notation. However, fixed notation is
+ // disabled, so report a scientific notation error.
+ return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+ }
+ // Otherwise, we will be ignoring the 'e'.
+ p = location_of_e;
+ } else {
+ while ((p != pend) && is_integer(*p)) {
+ uint8_t digit = uint8_t(*p - UC('0'));
+ if (exp_number < 0x10000000) {
+ exp_number = 10 * exp_number + digit;
+ }
+ ++p;
+ }
+ if (neg_exp) {
+ exp_number = -exp_number;
+ }
+ exponent += exp_number;
+ }
+ } else {
+ // If it scientific and not fixed, we have to bail out.
+ if (uint64_t(fmt & chars_format::scientific) &&
+ !uint64_t(fmt & chars_format::fixed)) {
+ return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+ }
+ }
+ answer.lastmatch = p;
+ answer.valid = true;
+
+ // If we frequently had to deal with long strings of digits,
+ // we could extend our code by using a 128-bit integer instead
+ // of a 64-bit integer. However, this is uncommon.
+ //
+ // We can deal with up to 19 digits.
+ if (digit_count > 19) { // this is uncommon
+ // It is possible that the integer had an overflow.
+ // We have to handle the case where we have 0.0000somenumber.
+ // We need to be mindful of the case where we only have zeroes...
+ // E.g., 0.000000000...000.
+ UC const *start = start_digits;
+ while ((start != pend) && (*start == UC('0') || *start == decimal_point)) {
+ if (*start == UC('0')) {
+ digit_count--;
+ }
+ start++;
+ }
+
+ if (digit_count > 19) {
+ answer.too_many_digits = true;
+ // The truncation recompute below reads the integer/fraction spans. When
+ // store_spans is false we didn't materialize them, so just flag
+ // too_many_digits; the caller re-parses with store_spans=true to obtain
+ // the corrected mantissa/exponent before taking the slow path.
+ if (store_spans) {
+ // Let us start again, this time, avoiding overflows.
+ // We don't need to call if is_integer, since we use the
+ // pre-tokenized spans from above.
+ i = 0;
+ p = answer.integer.ptr;
+ UC const *int_end = p + answer.integer.len();
+ uint64_t const minimal_nineteen_digit_integer{1000000000000000000};
+ while ((i < minimal_nineteen_digit_integer) && (p != int_end)) {
+ i = i * 10 + uint64_t(*p - UC('0'));
+ ++p;
+ }
+ if (i >= minimal_nineteen_digit_integer) { // We have a big integer
+ exponent = end_of_integer_part - p + exp_number;
+ } else { // We have a value with a fractional component.
+ p = answer.fraction.ptr;
+ UC const *frac_end = p + answer.fraction.len();
+ while ((i < minimal_nineteen_digit_integer) && (p != frac_end)) {
+ i = i * 10 + uint64_t(*p - UC('0'));
+ ++p;
+ }
+ exponent = answer.fraction.ptr - p + exp_number;
+ }
+ // We have now corrected both exponent and i, to a truncated value
+ }
+ }
+ }
+ answer.exponent = exponent;
+ answer.mantissa = i;
+ return answer;
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_int_string(UC const *p, UC const *pend, T &value,
+ parse_options_t<UC> options) {
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ int const base = options.base;
+
+ from_chars_result_t<UC> answer;
+
+ UC const *const first = p;
+
+ bool const negative = (*p == UC('-'));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4127)
+#endif
+ if (!std::is_signed<T>::value && negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+ if ((*p == UC('-')) ||
+ (uint64_t(fmt & chars_format::allow_leading_plus) && (*p == UC('+')))) {
+ ++p;
+ }
+
+ UC const *const start_num = p;
+
+ while (p != pend && *p == UC('0')) {
+ ++p;
+ }
+
+ bool const has_leading_zeros = p > start_num;
+
+ UC const *const start_digits = p;
+
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+ (std::is_same<T, std::uint8_t>::value && sizeof(UC) == 1)) {
+ if (base == 10) {
+ const size_t len = static_cast<size_t>(pend - p);
+ if (len == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ uint32_t digits;
+
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ if (std::is_constant_evaluated()) {
+ uint8_t str[4]{};
+ for (size_t j = 0; j < 4 && j < len; ++j) {
+ str[j] = static_cast<uint8_t>(p[j]);
+ }
+ digits = std::bit_cast<uint32_t>(str);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+ digits = byteswap_32(digits);
+#endif
+ }
+#else
+ if (false) {
+ }
+#endif
+ else if (len >= 4) {
+ ::memcpy(&digits, p, 4);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+ digits = byteswap_32(digits);
+#endif
+ } else {
+ uint32_t b0 = static_cast<uint8_t>(p[0]);
+ uint32_t b1 = (len > 1) ? static_cast<uint8_t>(p[1]) : 0xFFu;
+ uint32_t b2 = (len > 2) ? static_cast<uint8_t>(p[2]) : 0xFFu;
+ uint32_t b3 = 0xFFu;
+ digits = b0 | (b1 << 8) | (b2 << 16) | (b3 << 24);
+ }
+
+ uint32_t magic =
+ ((digits + 0x46464646u) | (digits - 0x30303030u)) & 0x80808080u;
+ uint32_t tz =
+ static_cast<uint32_t>(countr_zero_32(magic)); // 7, 15, 23, 31, or 32
+ uint32_t nd = (tz == 32) ? 4 : (tz >> 3);
+ nd = static_cast<uint32_t>(nd < len ? nd : len);
+ if (nd == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ return answer;
+ }
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+ if (nd > 3) {
+ const UC *q = p + nd;
+ size_t rem = len - nd;
+ while (rem) {
+ if (*q < UC('0') || *q > UC('9'))
+ break;
+ ++q;
+ --rem;
+ }
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = q;
+ return answer;
+ }
+
+ digits ^= 0x30303030u;
+ digits <<= ((4 - nd) * 8);
+
+ uint32_t check = ((digits >> 24) & 0xff) | ((digits >> 8) & 0xff00) |
+ ((digits << 8) & 0xff0000);
+ if (check > 0x00020505) {
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = p + nd;
+ return answer;
+ }
+ value = static_cast<uint8_t>((0x640a01 * digits) >> 24);
+ answer.ec = std::errc();
+ answer.ptr = p + nd;
+ return answer;
+ }
+ }
+
+ SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+ (std::is_same<T, std::uint16_t>::value && sizeof(UC) == 1)) {
+ if (base == 10) {
+ const size_t len = size_t(pend - p);
+ if (len == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ if (len >= 4) {
+ uint32_t digits = read4_to_u32(p);
+ if (is_made_of_four_digits_fast(digits)) {
+ uint32_t v = parse_four_digits_unrolled(digits);
+ if (len >= 5 && is_integer(p[4])) {
+ v = v * 10 + uint32_t(p[4] - '0');
+ if (len >= 6 && is_integer(p[5])) {
+ answer.ec = std::errc::result_out_of_range;
+ const UC *q = p + 5;
+ while (q != pend && is_integer(*q)) {
+ q++;
+ }
+ answer.ptr = q;
+ return answer;
+ }
+ if (v > 65535) {
+ answer.ec = std::errc::result_out_of_range;
+ answer.ptr = p + 5;
+ return answer;
+ }
+ value = uint16_t(v);
+ answer.ec = std::errc();
+ answer.ptr = p + 5;
+ return answer;
+ }
+ // 4 digits
+ value = uint16_t(v);
+ answer.ec = std::errc();
+ answer.ptr = p + 4;
+ return answer;
+ }
+ }
+ }
+ }
+
+ uint64_t i = 0;
+ if (base == 10) {
+ loop_parse_if_eight_digits(p, pend, i); // use SIMD if possible
+ }
+ while (p != pend) {
+ uint8_t digit = ch_to_digit(*p);
+ if (digit >= base) {
+ break;
+ }
+ i = uint64_t(base) * i + digit; // might overflow, check this later
+ p++;
+ }
+
+ size_t digit_count = size_t(p - start_digits);
+
+ if (digit_count == 0) {
+ if (has_leading_zeros) {
+ value = 0;
+ answer.ec = std::errc();
+ answer.ptr = p;
+ } else {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ }
+ return answer;
+ }
+
+ answer.ptr = p;
+
+ // check u64 overflow
+ size_t max_digits = max_digits_u64(base);
+ if (digit_count > max_digits) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ // this check can be eliminated for all other types, but they will all require
+ // a max_digits(base) equivalent
+ if (digit_count == max_digits) {
+ // At the max_digits boundary the accumulator `i` may have wrapped around
+ // 2^64. A plain `i < min_safe_u64(base)` test is not sufficient: for any
+ // base whose max_digits-length range exceeds 2^64 (base 10 reaches
+ // ~5.4 * 2^64 at 20 digits) the value can wrap a whole multiple of 2^64 and
+ // land back above min_safe, slipping through. Decide exactly in O(1) using
+ // the leading digit, following the approach used in simdjson:
+ // ms == min_safe_u64(base) == base^(max_digits-1), the smallest
+ // max_digits-length value.
+ // dmax == the largest leading digit whose number can still fit in u64.
+ // The leading-digit band [d*ms, (d+1)*ms) has width ms < 2^64, so within
+ // the single band where d == dmax the value straddles 2^64 at most once,
+ // and a single threshold separates wrapped from non-wrapped values. A
+ // leading digit above dmax always overflows; below dmax always fits.
+ uint64_t const ms = min_safe_u64(base);
+ uint64_t const dmax = (std::numeric_limits<uint64_t>::max)() / ms;
+ uint64_t const lead = ch_to_digit(*start_digits);
+ if (lead > dmax || (lead == dmax && i < dmax * ms)) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ }
+
+ // check other types overflow
+ if (!std::is_same<T, uint64_t>::value) {
+ if (i > uint64_t((std::numeric_limits<T>::max)()) + uint64_t(negative)) {
+ answer.ec = std::errc::result_out_of_range;
+ return answer;
+ }
+ }
+
+ if (negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4146)
+#endif
+ // this weird workaround is required because:
+ // - converting unsigned to signed when its value is greater than signed max
+ // is UB pre-C++23.
+ // - reinterpret_casting (~i + 1) would work, but it is not constexpr
+ // this is always optimized into a neg instruction (note: T is an integer
+ // type)
+ value = T(-(std::numeric_limits<T>::max)() -
+ T(i - uint64_t((std::numeric_limits<T>::max)())));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+ } else {
+ value = T(i);
+ }
+
+ answer.ec = std::errc();
+ return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_TABLE_H
+#define SIMDJSON_FASTFLOAT_FAST_TABLE_H
+
+#include <cstdint>
+
+namespace simdjson_fast_float {
+
+/**
+ * When mapping numbers from decimal to binary,
+ * we go from w * 10^q to m * 2^p but we have
+ * 10^q = 5^q * 2^q, so effectively
+ * we are trying to match
+ * w * 2^q * 5^q to m * 2^p. Thus the powers of two
+ * are not a concern since they can be represented
+ * exactly using the binary notation, only the powers of five
+ * affect the binary significand.
+ */
+
+/**
+ * The smallest non-zero float (binary64) is 2^-1074.
+ * We take as input numbers of the form w x 10^q where w < 2^64.
+ * We have that w * 10^-343 < 2^(64-344) 5^-343 < 2^-1076.
+ * However, we have that
+ * (2^64-1) * 10^-342 = (2^64-1) * 2^-342 * 5^-342 > 2^-1074.
+ * Thus it is possible for a number of the form w * 10^-342 where
+ * w is a 64-bit value to be a non-zero floating-point number.
+ *********
+ * Any number of form w * 10^309 where w>= 1 is going to be
+ * infinite in binary64 so we never need to worry about powers
+ * of 5 greater than 308.
+ */
+template <class unused = void> struct powers_template {
+
+ constexpr static int smallest_power_of_five =
+ binary_format<double>::smallest_power_of_ten();
+ constexpr static int largest_power_of_five =
+ binary_format<double>::largest_power_of_ten();
+ constexpr static int number_of_entries =
+ 2 * (largest_power_of_five - smallest_power_of_five + 1);
+ // Powers of five from 5^-342 all the way to 5^308 rounded toward one.
+ constexpr static uint64_t power_of_five_128[number_of_entries] = {
+ 0xeef453d6923bd65a, 0x113faa2906a13b3f,
+ 0x9558b4661b6565f8, 0x4ac7ca59a424c507,
+ 0xbaaee17fa23ebf76, 0x5d79bcf00d2df649,
+ 0xe95a99df8ace6f53, 0xf4d82c2c107973dc,
+ 0x91d8a02bb6c10594, 0x79071b9b8a4be869,
+ 0xb64ec836a47146f9, 0x9748e2826cdee284,
+ 0xe3e27a444d8d98b7, 0xfd1b1b2308169b25,
+ 0x8e6d8c6ab0787f72, 0xfe30f0f5e50e20f7,
+ 0xb208ef855c969f4f, 0xbdbd2d335e51a935,
+ 0xde8b2b66b3bc4723, 0xad2c788035e61382,
+ 0x8b16fb203055ac76, 0x4c3bcb5021afcc31,
+ 0xaddcb9e83c6b1793, 0xdf4abe242a1bbf3d,
+ 0xd953e8624b85dd78, 0xd71d6dad34a2af0d,
+ 0x87d4713d6f33aa6b, 0x8672648c40e5ad68,
+ 0xa9c98d8ccb009506, 0x680efdaf511f18c2,
+ 0xd43bf0effdc0ba48, 0x212bd1b2566def2,
+ 0x84a57695fe98746d, 0x14bb630f7604b57,
+ 0xa5ced43b7e3e9188, 0x419ea3bd35385e2d,
+ 0xcf42894a5dce35ea, 0x52064cac828675b9,
+ 0x818995ce7aa0e1b2, 0x7343efebd1940993,
+ 0xa1ebfb4219491a1f, 0x1014ebe6c5f90bf8,
+ 0xca66fa129f9b60a6, 0xd41a26e077774ef6,
+ 0xfd00b897478238d0, 0x8920b098955522b4,
+ 0x9e20735e8cb16382, 0x55b46e5f5d5535b0,
+ 0xc5a890362fddbc62, 0xeb2189f734aa831d,
+ 0xf712b443bbd52b7b, 0xa5e9ec7501d523e4,
+ 0x9a6bb0aa55653b2d, 0x47b233c92125366e,
+ 0xc1069cd4eabe89f8, 0x999ec0bb696e840a,
+ 0xf148440a256e2c76, 0xc00670ea43ca250d,
+ 0x96cd2a865764dbca, 0x380406926a5e5728,
+ 0xbc807527ed3e12bc, 0xc605083704f5ecf2,
+ 0xeba09271e88d976b, 0xf7864a44c633682e,
+ 0x93445b8731587ea3, 0x7ab3ee6afbe0211d,
+ 0xb8157268fdae9e4c, 0x5960ea05bad82964,
+ 0xe61acf033d1a45df, 0x6fb92487298e33bd,
+ 0x8fd0c16206306bab, 0xa5d3b6d479f8e056,
+ 0xb3c4f1ba87bc8696, 0x8f48a4899877186c,
+ 0xe0b62e2929aba83c, 0x331acdabfe94de87,
+ 0x8c71dcd9ba0b4925, 0x9ff0c08b7f1d0b14,
+ 0xaf8e5410288e1b6f, 0x7ecf0ae5ee44dd9,
+ 0xdb71e91432b1a24a, 0xc9e82cd9f69d6150,
+ 0x892731ac9faf056e, 0xbe311c083a225cd2,
+ 0xab70fe17c79ac6ca, 0x6dbd630a48aaf406,
+ 0xd64d3d9db981787d, 0x92cbbccdad5b108,
+ 0x85f0468293f0eb4e, 0x25bbf56008c58ea5,
+ 0xa76c582338ed2621, 0xaf2af2b80af6f24e,
+ 0xd1476e2c07286faa, 0x1af5af660db4aee1,
+ 0x82cca4db847945ca, 0x50d98d9fc890ed4d,
+ 0xa37fce126597973c, 0xe50ff107bab528a0,
+ 0xcc5fc196fefd7d0c, 0x1e53ed49a96272c8,
+ 0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7a,
+ 0x9faacf3df73609b1, 0x77b191618c54e9ac,
+ 0xc795830d75038c1d, 0xd59df5b9ef6a2417,
+ 0xf97ae3d0d2446f25, 0x4b0573286b44ad1d,
+ 0x9becce62836ac577, 0x4ee367f9430aec32,
+ 0xc2e801fb244576d5, 0x229c41f793cda73f,
+ 0xf3a20279ed56d48a, 0x6b43527578c1110f,
+ 0x9845418c345644d6, 0x830a13896b78aaa9,
+ 0xbe5691ef416bd60c, 0x23cc986bc656d553,
+ 0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa8,
+ 0x94b3a202eb1c3f39, 0x7bf7d71432f3d6a9,
+ 0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc53,
+ 0xe858ad248f5c22c9, 0xd1b3400f8f9cff68,
+ 0x91376c36d99995be, 0x23100809b9c21fa1,
+ 0xb58547448ffffb2d, 0xabd40a0c2832a78a,
+ 0xe2e69915b3fff9f9, 0x16c90c8f323f516c,
+ 0x8dd01fad907ffc3b, 0xae3da7d97f6792e3,
+ 0xb1442798f49ffb4a, 0x99cd11cfdf41779c,
+ 0xdd95317f31c7fa1d, 0x40405643d711d583,
+ 0x8a7d3eef7f1cfc52, 0x482835ea666b2572,
+ 0xad1c8eab5ee43b66, 0xda3243650005eecf,
+ 0xd863b256369d4a40, 0x90bed43e40076a82,
+ 0x873e4f75e2224e68, 0x5a7744a6e804a291,
+ 0xa90de3535aaae202, 0x711515d0a205cb36,
+ 0xd3515c2831559a83, 0xd5a5b44ca873e03,
+ 0x8412d9991ed58091, 0xe858790afe9486c2,
+ 0xa5178fff668ae0b6, 0x626e974dbe39a872,
+ 0xce5d73ff402d98e3, 0xfb0a3d212dc8128f,
+ 0x80fa687f881c7f8e, 0x7ce66634bc9d0b99,
+ 0xa139029f6a239f72, 0x1c1fffc1ebc44e80,
+ 0xc987434744ac874e, 0xa327ffb266b56220,
+ 0xfbe9141915d7a922, 0x4bf1ff9f0062baa8,
+ 0x9d71ac8fada6c9b5, 0x6f773fc3603db4a9,
+ 0xc4ce17b399107c22, 0xcb550fb4384d21d3,
+ 0xf6019da07f549b2b, 0x7e2a53a146606a48,
+ 0x99c102844f94e0fb, 0x2eda7444cbfc426d,
+ 0xc0314325637a1939, 0xfa911155fefb5308,
+ 0xf03d93eebc589f88, 0x793555ab7eba27ca,
+ 0x96267c7535b763b5, 0x4bc1558b2f3458de,
+ 0xbbb01b9283253ca2, 0x9eb1aaedfb016f16,
+ 0xea9c227723ee8bcb, 0x465e15a979c1cadc,
+ 0x92a1958a7675175f, 0xbfacd89ec191ec9,
+ 0xb749faed14125d36, 0xcef980ec671f667b,
+ 0xe51c79a85916f484, 0x82b7e12780e7401a,
+ 0x8f31cc0937ae58d2, 0xd1b2ecb8b0908810,
+ 0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa15,
+ 0xdfbdcece67006ac9, 0x67a791e093e1d49a,
+ 0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e0,
+ 0xaecc49914078536d, 0x58fae9f773886e18,
+ 0xda7f5bf590966848, 0xaf39a475506a899e,
+ 0x888f99797a5e012d, 0x6d8406c952429603,
+ 0xaab37fd7d8f58178, 0xc8e5087ba6d33b83,
+ 0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a64,
+ 0x855c3be0a17fcd26, 0x5cf2eea09a55067f,
+ 0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481e,
+ 0xd0601d8efc57b08b, 0xf13b94daf124da26,
+ 0x823c12795db6ce57, 0x76c53d08d6b70858,
+ 0xa2cb1717b52481ed, 0x54768c4b0c64ca6e,
+ 0xcb7ddcdda26da268, 0xa9942f5dcf7dfd09,
+ 0xfe5d54150b090b02, 0xd3f93b35435d7c4c,
+ 0x9efa548d26e5a6e1, 0xc47bc5014a1a6daf,
+ 0xc6b8e9b0709f109a, 0x359ab6419ca1091b,
+ 0xf867241c8cc6d4c0, 0xc30163d203c94b62,
+ 0x9b407691d7fc44f8, 0x79e0de63425dcf1d,
+ 0xc21094364dfb5636, 0x985915fc12f542e4,
+ 0xf294b943e17a2bc4, 0x3e6f5b7b17b2939d,
+ 0x979cf3ca6cec5b5a, 0xa705992ceecf9c42,
+ 0xbd8430bd08277231, 0x50c6ff782a838353,
+ 0xece53cec4a314ebd, 0xa4f8bf5635246428,
+ 0x940f4613ae5ed136, 0x871b7795e136be99,
+ 0xb913179899f68584, 0x28e2557b59846e3f,
+ 0xe757dd7ec07426e5, 0x331aeada2fe589cf,
+ 0x9096ea6f3848984f, 0x3ff0d2c85def7621,
+ 0xb4bca50b065abe63, 0xfed077a756b53a9,
+ 0xe1ebce4dc7f16dfb, 0xd3e8495912c62894,
+ 0x8d3360f09cf6e4bd, 0x64712dd7abbbd95c,
+ 0xb080392cc4349dec, 0xbd8d794d96aacfb3,
+ 0xdca04777f541c567, 0xecf0d7a0fc5583a0,
+ 0x89e42caaf9491b60, 0xf41686c49db57244,
+ 0xac5d37d5b79b6239, 0x311c2875c522ced5,
+ 0xd77485cb25823ac7, 0x7d633293366b828b,
+ 0x86a8d39ef77164bc, 0xae5dff9c02033197,
+ 0xa8530886b54dbdeb, 0xd9f57f830283fdfc,
+ 0xd267caa862a12d66, 0xd072df63c324fd7b,
+ 0x8380dea93da4bc60, 0x4247cb9e59f71e6d,
+ 0xa46116538d0deb78, 0x52d9be85f074e608,
+ 0xcd795be870516656, 0x67902e276c921f8b,
+ 0x806bd9714632dff6, 0xba1cd8a3db53b6,
+ 0xa086cfcd97bf97f3, 0x80e8a40eccd228a4,
+ 0xc8a883c0fdaf7df0, 0x6122cd128006b2cd,
+ 0xfad2a4b13d1b5d6c, 0x796b805720085f81,
+ 0x9cc3a6eec6311a63, 0xcbe3303674053bb0,
+ 0xc3f490aa77bd60fc, 0xbedbfc4411068a9c,
+ 0xf4f1b4d515acb93b, 0xee92fb5515482d44,
+ 0x991711052d8bf3c5, 0x751bdd152d4d1c4a,
+ 0xbf5cd54678eef0b6, 0xd262d45a78a0635d,
+ 0xef340a98172aace4, 0x86fb897116c87c34,
+ 0x9580869f0e7aac0e, 0xd45d35e6ae3d4da0,
+ 0xbae0a846d2195712, 0x8974836059cca109,
+ 0xe998d258869facd7, 0x2bd1a438703fc94b,
+ 0x91ff83775423cc06, 0x7b6306a34627ddcf,
+ 0xb67f6455292cbf08, 0x1a3bc84c17b1d542,
+ 0xe41f3d6a7377eeca, 0x20caba5f1d9e4a93,
+ 0x8e938662882af53e, 0x547eb47b7282ee9c,
+ 0xb23867fb2a35b28d, 0xe99e619a4f23aa43,
+ 0xdec681f9f4c31f31, 0x6405fa00e2ec94d4,
+ 0x8b3c113c38f9f37e, 0xde83bc408dd3dd04,
+ 0xae0b158b4738705e, 0x9624ab50b148d445,
+ 0xd98ddaee19068c76, 0x3badd624dd9b0957,
+ 0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d6,
+ 0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4c,
+ 0xd47487cc8470652b, 0x7647c3200069671f,
+ 0x84c8d4dfd2c63f3b, 0x29ecd9f40041e073,
+ 0xa5fb0a17c777cf09, 0xf468107100525890,
+ 0xcf79cc9db955c2cc, 0x7182148d4066eeb4,
+ 0x81ac1fe293d599bf, 0xc6f14cd848405530,
+ 0xa21727db38cb002f, 0xb8ada00e5a506a7c,
+ 0xca9cf1d206fdc03b, 0xa6d90811f0e4851c,
+ 0xfd442e4688bd304a, 0x908f4a166d1da663,
+ 0x9e4a9cec15763e2e, 0x9a598e4e043287fe,
+ 0xc5dd44271ad3cdba, 0x40eff1e1853f29fd,
+ 0xf7549530e188c128, 0xd12bee59e68ef47c,
+ 0x9a94dd3e8cf578b9, 0x82bb74f8301958ce,
+ 0xc13a148e3032d6e7, 0xe36a52363c1faf01,
+ 0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac1,
+ 0x96f5600f15a7b7e5, 0x29ab103a5ef8c0b9,
+ 0xbcb2b812db11a5de, 0x7415d448f6b6f0e7,
+ 0xebdf661791d60f56, 0x111b495b3464ad21,
+ 0x936b9fcebb25c995, 0xcab10dd900beec34,
+ 0xb84687c269ef3bfb, 0x3d5d514f40eea742,
+ 0xe65829b3046b0afa, 0xcb4a5a3112a5112,
+ 0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ab,
+ 0xb3f4e093db73a093, 0x59ed216765690f56,
+ 0xe0f218b8d25088b8, 0x306869c13ec3532c,
+ 0x8c974f7383725573, 0x1e414218c73a13fb,
+ 0xafbd2350644eeacf, 0xe5d1929ef90898fa,
+ 0xdbac6c247d62a583, 0xdf45f746b74abf39,
+ 0x894bc396ce5da772, 0x6b8bba8c328eb783,
+ 0xab9eb47c81f5114f, 0x66ea92f3f326564,
+ 0xd686619ba27255a2, 0xc80a537b0efefebd,
+ 0x8613fd0145877585, 0xbd06742ce95f5f36,
+ 0xa798fc4196e952e7, 0x2c48113823b73704,
+ 0xd17f3b51fca3a7a0, 0xf75a15862ca504c5,
+ 0x82ef85133de648c4, 0x9a984d73dbe722fb,
+ 0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebba,
+ 0xcc963fee10b7d1b3, 0x318df905079926a8,
+ 0xffbbcfe994e5c61f, 0xfdf17746497f7052,
+ 0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa633,
+ 0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc0,
+ 0xf9bd690a1b68637b, 0x3dfdce7aa3c673b0,
+ 0x9c1661a651213e2d, 0x6bea10ca65c084e,
+ 0xc31bfa0fe5698db8, 0x486e494fcff30a62,
+ 0xf3e2f893dec3f126, 0x5a89dba3c3efccfa,
+ 0x986ddb5c6b3a76b7, 0xf89629465a75e01c,
+ 0xbe89523386091465, 0xf6bbb397f1135823,
+ 0xee2ba6c0678b597f, 0x746aa07ded582e2c,
+ 0x94db483840b717ef, 0xa8c2a44eb4571cdc,
+ 0xba121a4650e4ddeb, 0x92f34d62616ce413,
+ 0xe896a0d7e51e1566, 0x77b020baf9c81d17,
+ 0x915e2486ef32cd60, 0xace1474dc1d122e,
+ 0xb5b5ada8aaff80b8, 0xd819992132456ba,
+ 0xe3231912d5bf60e6, 0x10e1fff697ed6c69,
+ 0x8df5efabc5979c8f, 0xca8d3ffa1ef463c1,
+ 0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb2,
+ 0xddd0467c64bce4a0, 0xac7cb3f6d05ddbde,
+ 0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96b,
+ 0xad4ab7112eb3929d, 0x86c16c98d2c953c6,
+ 0xd89d64d57a607744, 0xe871c7bf077ba8b7,
+ 0x87625f056c7c4a8b, 0x11471cd764ad4972,
+ 0xa93af6c6c79b5d2d, 0xd598e40d3dd89bcf,
+ 0xd389b47879823479, 0x4aff1d108d4ec2c3,
+ 0x843610cb4bf160cb, 0xcedf722a585139ba,
+ 0xa54394fe1eedb8fe, 0xc2974eb4ee658828,
+ 0xce947a3da6a9273e, 0x733d226229feea32,
+ 0x811ccc668829b887, 0x806357d5a3f525f,
+ 0xa163ff802a3426a8, 0xca07c2dcb0cf26f7,
+ 0xc9bcff6034c13052, 0xfc89b393dd02f0b5,
+ 0xfc2c3f3841f17c67, 0xbbac2078d443ace2,
+ 0x9d9ba7832936edc0, 0xd54b944b84aa4c0d,
+ 0xc5029163f384a931, 0xa9e795e65d4df11,
+ 0xf64335bcf065d37d, 0x4d4617b5ff4a16d5,
+ 0x99ea0196163fa42e, 0x504bced1bf8e4e45,
+ 0xc06481fb9bcf8d39, 0xe45ec2862f71e1d6,
+ 0xf07da27a82c37088, 0x5d767327bb4e5a4c,
+ 0x964e858c91ba2655, 0x3a6a07f8d510f86f,
+ 0xbbe226efb628afea, 0x890489f70a55368b,
+ 0xeadab0aba3b2dbe5, 0x2b45ac74ccea842e,
+ 0x92c8ae6b464fc96f, 0x3b0b8bc90012929d,
+ 0xb77ada0617e3bbcb, 0x9ce6ebb40173744,
+ 0xe55990879ddcaabd, 0xcc420a6a101d0515,
+ 0x8f57fa54c2a9eab6, 0x9fa946824a12232d,
+ 0xb32df8e9f3546564, 0x47939822dc96abf9,
+ 0xdff9772470297ebd, 0x59787e2b93bc56f7,
+ 0x8bfbea76c619ef36, 0x57eb4edb3c55b65a,
+ 0xaefae51477a06b03, 0xede622920b6b23f1,
+ 0xdab99e59958885c4, 0xe95fab368e45eced,
+ 0x88b402f7fd75539b, 0x11dbcb0218ebb414,
+ 0xaae103b5fcd2a881, 0xd652bdc29f26a119,
+ 0xd59944a37c0752a2, 0x4be76d3346f0495f,
+ 0x857fcae62d8493a5, 0x6f70a4400c562ddb,
+ 0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb952,
+ 0xd097ad07a71f26b2, 0x7e2000a41346a7a7,
+ 0x825ecc24c873782f, 0x8ed400668c0c28c8,
+ 0xa2f67f2dfa90563b, 0x728900802f0f32fa,
+ 0xcbb41ef979346bca, 0x4f2b40a03ad2ffb9,
+ 0xfea126b7d78186bc, 0xe2f610c84987bfa8,
+ 0x9f24b832e6b0f436, 0xdd9ca7d2df4d7c9,
+ 0xc6ede63fa05d3143, 0x91503d1c79720dbb,
+ 0xf8a95fcf88747d94, 0x75a44c6397ce912a,
+ 0x9b69dbe1b548ce7c, 0xc986afbe3ee11aba,
+ 0xc24452da229b021b, 0xfbe85badce996168,
+ 0xf2d56790ab41c2a2, 0xfae27299423fb9c3,
+ 0x97c560ba6b0919a5, 0xdccd879fc967d41a,
+ 0xbdb6b8e905cb600f, 0x5400e987bbc1c920,
+ 0xed246723473e3813, 0x290123e9aab23b68,
+ 0x9436c0760c86e30b, 0xf9a0b6720aaf6521,
+ 0xb94470938fa89bce, 0xf808e40e8d5b3e69,
+ 0xe7958cb87392c2c2, 0xb60b1d1230b20e04,
+ 0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c2,
+ 0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af3,
+ 0xe2280b6c20dd5232, 0x25c6da63c38de1b0,
+ 0x8d590723948a535f, 0x579c487e5a38ad0e,
+ 0xb0af48ec79ace837, 0x2d835a9df0c6d851,
+ 0xdcdb1b2798182244, 0xf8e431456cf88e65,
+ 0x8a08f0f8bf0f156b, 0x1b8e9ecb641b58ff,
+ 0xac8b2d36eed2dac5, 0xe272467e3d222f3f,
+ 0xd7adf884aa879177, 0x5b0ed81dcc6abb0f,
+ 0x86ccbb52ea94baea, 0x98e947129fc2b4e9,
+ 0xa87fea27a539e9a5, 0x3f2398d747b36224,
+ 0xd29fe4b18e88640e, 0x8eec7f0d19a03aad,
+ 0x83a3eeeef9153e89, 0x1953cf68300424ac,
+ 0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd7,
+ 0xcdb02555653131b6, 0x3792f412cb06794d,
+ 0x808e17555f3ebf11, 0xe2bbd88bbee40bd0,
+ 0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec4,
+ 0xc8de047564d20a8b, 0xf245825a5a445275,
+ 0xfb158592be068d2e, 0xeed6e2f0f0d56712,
+ 0x9ced737bb6c4183d, 0x55464dd69685606b,
+ 0xc428d05aa4751e4c, 0xaa97e14c3c26b886,
+ 0xf53304714d9265df, 0xd53dd99f4b3066a8,
+ 0x993fe2c6d07b7fab, 0xe546a8038efe4029,
+ 0xbf8fdb78849a5f96, 0xde98520472bdd033,
+ 0xef73d256a5c0f77c, 0x963e66858f6d4440,
+ 0x95a8637627989aad, 0xdde7001379a44aa8,
+ 0xbb127c53b17ec159, 0x5560c018580d5d52,
+ 0xe9d71b689dde71af, 0xaab8f01e6e10b4a6,
+ 0x9226712162ab070d, 0xcab3961304ca70e8,
+ 0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d22,
+ 0xe45c10c42a2b3b05, 0x8cb89a7db77c506a,
+ 0x8eb98a7a9a5b04e3, 0x77f3608e92adb242,
+ 0xb267ed1940f1c61c, 0x55f038b237591ed3,
+ 0xdf01e85f912e37a3, 0x6b6c46dec52f6688,
+ 0x8b61313bbabce2c6, 0x2323ac4b3b3da015,
+ 0xae397d8aa96c1b77, 0xabec975e0a0d081a,
+ 0xd9c7dced53c72255, 0x96e7bd358c904a21,
+ 0x881cea14545c7575, 0x7e50d64177da2e54,
+ 0xaa242499697392d2, 0xdde50bd1d5d0b9e9,
+ 0xd4ad2dbfc3d07787, 0x955e4ec64b44e864,
+ 0x84ec3c97da624ab4, 0xbd5af13bef0b113e,
+ 0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58e,
+ 0xcfb11ead453994ba, 0x67de18eda5814af2,
+ 0x81ceb32c4b43fcf4, 0x80eacf948770ced7,
+ 0xa2425ff75e14fc31, 0xa1258379a94d028d,
+ 0xcad2f7f5359a3b3e, 0x96ee45813a04330,
+ 0xfd87b5f28300ca0d, 0x8bca9d6e188853fc,
+ 0x9e74d1b791e07e48, 0x775ea264cf55347e,
+ 0xc612062576589dda, 0x95364afe032a819e,
+ 0xf79687aed3eec551, 0x3a83ddbd83f52205,
+ 0x9abe14cd44753b52, 0xc4926a9672793543,
+ 0xc16d9a0095928a27, 0x75b7053c0f178294,
+ 0xf1c90080baf72cb1, 0x5324c68b12dd6339,
+ 0x971da05074da7bee, 0xd3f6fc16ebca5e04,
+ 0xbce5086492111aea, 0x88f4bb1ca6bcf585,
+ 0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6,
+ 0x9392ee8e921d5d07, 0x3aff322e62439fd0,
+ 0xb877aa3236a4b449, 0x9befeb9fad487c3,
+ 0xe69594bec44de15b, 0x4c2ebe687989a9b4,
+ 0x901d7cf73ab0acd9, 0xf9d37014bf60a11,
+ 0xb424dc35095cd80f, 0x538484c19ef38c95,
+ 0xe12e13424bb40e13, 0x2865a5f206b06fba,
+ 0x8cbccc096f5088cb, 0xf93f87b7442e45d4,
+ 0xafebff0bcb24aafe, 0xf78f69a51539d749,
+ 0xdbe6fecebdedd5be, 0xb573440e5a884d1c,
+ 0x89705f4136b4a597, 0x31680a88f8953031,
+ 0xabcc77118461cefc, 0xfdc20d2b36ba7c3e,
+ 0xd6bf94d5e57a42bc, 0x3d32907604691b4d,
+ 0x8637bd05af6c69b5, 0xa63f9a49c2c1b110,
+ 0xa7c5ac471b478423, 0xfcf80dc33721d54,
+ 0xd1b71758e219652b, 0xd3c36113404ea4a9,
+ 0x83126e978d4fdf3b, 0x645a1cac083126ea,
+ 0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4,
+ 0xcccccccccccccccc, 0xcccccccccccccccd,
+ 0x8000000000000000, 0x0,
+ 0xa000000000000000, 0x0,
+ 0xc800000000000000, 0x0,
+ 0xfa00000000000000, 0x0,
+ 0x9c40000000000000, 0x0,
+ 0xc350000000000000, 0x0,
+ 0xf424000000000000, 0x0,
+ 0x9896800000000000, 0x0,
+ 0xbebc200000000000, 0x0,
+ 0xee6b280000000000, 0x0,
+ 0x9502f90000000000, 0x0,
+ 0xba43b74000000000, 0x0,
+ 0xe8d4a51000000000, 0x0,
+ 0x9184e72a00000000, 0x0,
+ 0xb5e620f480000000, 0x0,
+ 0xe35fa931a0000000, 0x0,
+ 0x8e1bc9bf04000000, 0x0,
+ 0xb1a2bc2ec5000000, 0x0,
+ 0xde0b6b3a76400000, 0x0,
+ 0x8ac7230489e80000, 0x0,
+ 0xad78ebc5ac620000, 0x0,
+ 0xd8d726b7177a8000, 0x0,
+ 0x878678326eac9000, 0x0,
+ 0xa968163f0a57b400, 0x0,
+ 0xd3c21bcecceda100, 0x0,
+ 0x84595161401484a0, 0x0,
+ 0xa56fa5b99019a5c8, 0x0,
+ 0xcecb8f27f4200f3a, 0x0,
+ 0x813f3978f8940984, 0x4000000000000000,
+ 0xa18f07d736b90be5, 0x5000000000000000,
+ 0xc9f2c9cd04674ede, 0xa400000000000000,
+ 0xfc6f7c4045812296, 0x4d00000000000000,
+ 0x9dc5ada82b70b59d, 0xf020000000000000,
+ 0xc5371912364ce305, 0x6c28000000000000,
+ 0xf684df56c3e01bc6, 0xc732000000000000,
+ 0x9a130b963a6c115c, 0x3c7f400000000000,
+ 0xc097ce7bc90715b3, 0x4b9f100000000000,
+ 0xf0bdc21abb48db20, 0x1e86d40000000000,
+ 0x96769950b50d88f4, 0x1314448000000000,
+ 0xbc143fa4e250eb31, 0x17d955a000000000,
+ 0xeb194f8e1ae525fd, 0x5dcfab0800000000,
+ 0x92efd1b8d0cf37be, 0x5aa1cae500000000,
+ 0xb7abc627050305ad, 0xf14a3d9e40000000,
+ 0xe596b7b0c643c719, 0x6d9ccd05d0000000,
+ 0x8f7e32ce7bea5c6f, 0xe4820023a2000000,
+ 0xb35dbf821ae4f38b, 0xdda2802c8a800000,
+ 0xe0352f62a19e306e, 0xd50b2037ad200000,
+ 0x8c213d9da502de45, 0x4526f422cc340000,
+ 0xaf298d050e4395d6, 0x9670b12b7f410000,
+ 0xdaf3f04651d47b4c, 0x3c0cdd765f114000,
+ 0x88d8762bf324cd0f, 0xa5880a69fb6ac800,
+ 0xab0e93b6efee0053, 0x8eea0d047a457a00,
+ 0xd5d238a4abe98068, 0x72a4904598d6d880,
+ 0x85a36366eb71f041, 0x47a6da2b7f864750,
+ 0xa70c3c40a64e6c51, 0x999090b65f67d924,
+ 0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d,
+ 0x82818f1281ed449f, 0xbff8f10e7a8921a4,
+ 0xa321f2d7226895c7, 0xaff72d52192b6a0d,
+ 0xcbea6f8ceb02bb39, 0x9bf4f8a69f764490,
+ 0xfee50b7025c36a08, 0x2f236d04753d5b4,
+ 0x9f4f2726179a2245, 0x1d762422c946590,
+ 0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef5,
+ 0xf8ebad2b84e0d58b, 0xd2e0898765a7deb2,
+ 0x9b934c3b330c8577, 0x63cc55f49f88eb2f,
+ 0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fb,
+ 0xf316271c7fc3908a, 0x8bef464e3945ef7a,
+ 0x97edd871cfda3a56, 0x97758bf0e3cbb5ac,
+ 0xbde94e8e43d0c8ec, 0x3d52eeed1cbea317,
+ 0xed63a231d4c4fb27, 0x4ca7aaa863ee4bdd,
+ 0x945e455f24fb1cf8, 0x8fe8caa93e74ef6a,
+ 0xb975d6b6ee39e436, 0xb3e2fd538e122b44,
+ 0xe7d34c64a9c85d44, 0x60dbbca87196b616,
+ 0x90e40fbeea1d3a4a, 0xbc8955e946fe31cd,
+ 0xb51d13aea4a488dd, 0x6babab6398bdbe41,
+ 0xe264589a4dcdab14, 0xc696963c7eed2dd1,
+ 0x8d7eb76070a08aec, 0xfc1e1de5cf543ca2,
+ 0xb0de65388cc8ada8, 0x3b25a55f43294bcb,
+ 0xdd15fe86affad912, 0x49ef0eb713f39ebe,
+ 0x8a2dbf142dfcc7ab, 0x6e3569326c784337,
+ 0xacb92ed9397bf996, 0x49c2c37f07965404,
+ 0xd7e77a8f87daf7fb, 0xdc33745ec97be906,
+ 0x86f0ac99b4e8dafd, 0x69a028bb3ded71a3,
+ 0xa8acd7c0222311bc, 0xc40832ea0d68ce0c,
+ 0xd2d80db02aabd62b, 0xf50a3fa490c30190,
+ 0x83c7088e1aab65db, 0x792667c6da79e0fa,
+ 0xa4b8cab1a1563f52, 0x577001b891185938,
+ 0xcde6fd5e09abcf26, 0xed4c0226b55e6f86,
+ 0x80b05e5ac60b6178, 0x544f8158315b05b4,
+ 0xa0dc75f1778e39d6, 0x696361ae3db1c721,
+ 0xc913936dd571c84c, 0x3bc3a19cd1e38e9,
+ 0xfb5878494ace3a5f, 0x4ab48a04065c723,
+ 0x9d174b2dcec0e47b, 0x62eb0d64283f9c76,
+ 0xc45d1df942711d9a, 0x3ba5d0bd324f8394,
+ 0xf5746577930d6500, 0xca8f44ec7ee36479,
+ 0x9968bf6abbe85f20, 0x7e998b13cf4e1ecb,
+ 0xbfc2ef456ae276e8, 0x9e3fedd8c321a67e,
+ 0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101e,
+ 0x95d04aee3b80ece5, 0xbba1f1d158724a12,
+ 0xbb445da9ca61281f, 0x2a8a6e45ae8edc97,
+ 0xea1575143cf97226, 0xf52d09d71a3293bd,
+ 0x924d692ca61be758, 0x593c2626705f9c56,
+ 0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836c,
+ 0xe498f455c38b997a, 0xb6dfb9c0f956447,
+ 0x8edf98b59a373fec, 0x4724bd4189bd5eac,
+ 0xb2977ee300c50fe7, 0x58edec91ec2cb657,
+ 0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ed,
+ 0x8b865b215899f46c, 0xbd79e0d20082ee74,
+ 0xae67f1e9aec07187, 0xecd8590680a3aa11,
+ 0xda01ee641a708de9, 0xe80e6f4820cc9495,
+ 0x884134fe908658b2, 0x3109058d147fdcdd,
+ 0xaa51823e34a7eede, 0xbd4b46f0599fd415,
+ 0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91a,
+ 0x850fadc09923329e, 0x3e2cf6bc604ddb0,
+ 0xa6539930bf6bff45, 0x84db8346b786151c,
+ 0xcfe87f7cef46ff16, 0xe612641865679a63,
+ 0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07e,
+ 0xa26da3999aef7749, 0xe3be5e330f38f09d,
+ 0xcb090c8001ab551c, 0x5cadf5bfd3072cc5,
+ 0xfdcb4fa002162a63, 0x73d9732fc7c8f7f6,
+ 0x9e9f11c4014dda7e, 0x2867e7fddcdd9afa,
+ 0xc646d63501a1511d, 0xb281e1fd541501b8,
+ 0xf7d88bc24209a565, 0x1f225a7ca91a4226,
+ 0x9ae757596946075f, 0x3375788de9b06958,
+ 0xc1a12d2fc3978937, 0x52d6b1641c83ae,
+ 0xf209787bb47d6b84, 0xc0678c5dbd23a49a,
+ 0x9745eb4d50ce6332, 0xf840b7ba963646e0,
+ 0xbd176620a501fbff, 0xb650e5a93bc3d898,
+ 0xec5d3fa8ce427aff, 0xa3e51f138ab4cebe,
+ 0x93ba47c980e98cdf, 0xc66f336c36b10137,
+ 0xb8a8d9bbe123f017, 0xb80b0047445d4184,
+ 0xe6d3102ad96cec1d, 0xa60dc059157491e5,
+ 0x9043ea1ac7e41392, 0x87c89837ad68db2f,
+ 0xb454e4a179dd1877, 0x29babe4598c311fb,
+ 0xe16a1dc9d8545e94, 0xf4296dd6fef3d67a,
+ 0x8ce2529e2734bb1d, 0x1899e4a65f58660c,
+ 0xb01ae745b101e9e4, 0x5ec05dcff72e7f8f,
+ 0xdc21a1171d42645d, 0x76707543f4fa1f73,
+ 0x899504ae72497eba, 0x6a06494a791c53a8,
+ 0xabfa45da0edbde69, 0x487db9d17636892,
+ 0xd6f8d7509292d603, 0x45a9d2845d3c42b6,
+ 0x865b86925b9bc5c2, 0xb8a2392ba45a9b2,
+ 0xa7f26836f282b732, 0x8e6cac7768d7141e,
+ 0xd1ef0244af2364ff, 0x3207d795430cd926,
+ 0x8335616aed761f1f, 0x7f44e6bd49e807b8,
+ 0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a6,
+ 0xcd036837130890a1, 0x36dba887c37a8c0f,
+ 0x802221226be55a64, 0xc2494954da2c9789,
+ 0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6c,
+ 0xc83553c5c8965d3d, 0x6f92829494e5acc7,
+ 0xfa42a8b73abbf48c, 0xcb772339ba1f17f9,
+ 0x9c69a97284b578d7, 0xff2a760414536efb,
+ 0xc38413cf25e2d70d, 0xfef5138519684aba,
+ 0xf46518c2ef5b8cd1, 0x7eb258665fc25d69,
+ 0x98bf2f79d5993802, 0xef2f773ffbd97a61,
+ 0xbeeefb584aff8603, 0xaafb550ffacfd8fa,
+ 0xeeaaba2e5dbf6784, 0x95ba2a53f983cf38,
+ 0x952ab45cfa97a0b2, 0xdd945a747bf26183,
+ 0xba756174393d88df, 0x94f971119aeef9e4,
+ 0xe912b9d1478ceb17, 0x7a37cd5601aab85d,
+ 0x91abb422ccb812ee, 0xac62e055c10ab33a,
+ 0xb616a12b7fe617aa, 0x577b986b314d6009,
+ 0xe39c49765fdf9d94, 0xed5a7e85fda0b80b,
+ 0x8e41ade9fbebc27d, 0x14588f13be847307,
+ 0xb1d219647ae6b31c, 0x596eb2d8ae258fc8,
+ 0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bb,
+ 0x8aec23d680043bee, 0x25de7bb9480d5854,
+ 0xada72ccc20054ae9, 0xaf561aa79a10ae6a,
+ 0xd910f7ff28069da4, 0x1b2ba1518094da04,
+ 0x87aa9aff79042286, 0x90fb44d2f05d0842,
+ 0xa99541bf57452b28, 0x353a1607ac744a53,
+ 0xd3fa922f2d1675f2, 0x42889b8997915ce8,
+ 0x847c9b5d7c2e09b7, 0x69956135febada11,
+ 0xa59bc234db398c25, 0x43fab9837e699095,
+ 0xcf02b2c21207ef2e, 0x94f967e45e03f4bb,
+ 0x8161afb94b44f57d, 0x1d1be0eebac278f5,
+ 0xa1ba1ba79e1632dc, 0x6462d92a69731732,
+ 0xca28a291859bbf93, 0x7d7b8f7503cfdcfe,
+ 0xfcb2cb35e702af78, 0x5cda735244c3d43e,
+ 0x9defbf01b061adab, 0x3a0888136afa64a7,
+ 0xc56baec21c7a1916, 0x88aaa1845b8fdd0,
+ 0xf6c69a72a3989f5b, 0x8aad549e57273d45,
+ 0x9a3c2087a63f6399, 0x36ac54e2f678864b,
+ 0xc0cb28a98fcf3c7f, 0x84576a1bb416a7dd,
+ 0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d5,
+ 0x969eb7c47859e743, 0x9f644ae5a4b1b325,
+ 0xbc4665b596706114, 0x873d5d9f0dde1fee,
+ 0xeb57ff22fc0c7959, 0xa90cb506d155a7ea,
+ 0x9316ff75dd87cbd8, 0x9a7f12442d588f2,
+ 0xb7dcbf5354e9bece, 0xc11ed6d538aeb2f,
+ 0xe5d3ef282a242e81, 0x8f1668c8a86da5fa,
+ 0x8fa475791a569d10, 0xf96e017d694487bc,
+ 0xb38d92d760ec4455, 0x37c981dcc395a9ac,
+ 0xe070f78d3927556a, 0x85bbe253f47b1417,
+ 0x8c469ab843b89562, 0x93956d7478ccec8e,
+ 0xaf58416654a6babb, 0x387ac8d1970027b2,
+ 0xdb2e51bfe9d0696a, 0x6997b05fcc0319e,
+ 0x88fcf317f22241e2, 0x441fece3bdf81f03,
+ 0xab3c2fddeeaad25a, 0xd527e81cad7626c3,
+ 0xd60b3bd56a5586f1, 0x8a71e223d8d3b074,
+ 0x85c7056562757456, 0xf6872d5667844e49,
+ 0xa738c6bebb12d16c, 0xb428f8ac016561db,
+ 0xd106f86e69d785c7, 0xe13336d701beba52,
+ 0x82a45b450226b39c, 0xecc0024661173473,
+ 0xa34d721642b06084, 0x27f002d7f95d0190,
+ 0xcc20ce9bd35c78a5, 0x31ec038df7b441f4,
+ 0xff290242c83396ce, 0x7e67047175a15271,
+ 0x9f79a169bd203e41, 0xf0062c6e984d386,
+ 0xc75809c42c684dd1, 0x52c07b78a3e60868,
+ 0xf92e0c3537826145, 0xa7709a56ccdf8a82,
+ 0x9bbcc7a142b17ccb, 0x88a66076400bb691,
+ 0xc2abf989935ddbfe, 0x6acff893d00ea435,
+ 0xf356f7ebf83552fe, 0x583f6b8c4124d43,
+ 0x98165af37b2153de, 0xc3727a337a8b704a,
+ 0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5c,
+ 0xeda2ee1c7064130c, 0x1162def06f79df73,
+ 0x9485d4d1c63e8be7, 0x8addcb5645ac2ba8,
+ 0xb9a74a0637ce2ee1, 0x6d953e2bd7173692,
+ 0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0437,
+ 0x910ab1d4db9914a0, 0x1d9c9892400a22a2,
+ 0xb54d5e4a127f59c8, 0x2503beb6d00cab4b,
+ 0xe2a0b5dc971f303a, 0x2e44ae64840fd61d,
+ 0x8da471a9de737e24, 0x5ceaecfed289e5d2,
+ 0xb10d8e1456105dad, 0x7425a83e872c5f47,
+ 0xdd50f1996b947518, 0xd12f124e28f77719,
+ 0x8a5296ffe33cc92f, 0x82bd6b70d99aaa6f,
+ 0xace73cbfdc0bfb7b, 0x636cc64d1001550b,
+ 0xd8210befd30efa5a, 0x3c47f7e05401aa4e,
+ 0x8714a775e3e95c78, 0x65acfaec34810a71,
+ 0xa8d9d1535ce3b396, 0x7f1839a741a14d0d,
+ 0xd31045a8341ca07c, 0x1ede48111209a050,
+ 0x83ea2b892091e44d, 0x934aed0aab460432,
+ 0xa4e4b66b68b65d60, 0xf81da84d5617853f,
+ 0xce1de40642e3f4b9, 0x36251260ab9d668e,
+ 0x80d2ae83e9ce78f3, 0xc1d72b7c6b426019,
+ 0xa1075a24e4421730, 0xb24cf65b8612f81f,
+ 0xc94930ae1d529cfc, 0xdee033f26797b627,
+ 0xfb9b7cd9a4a7443c, 0x169840ef017da3b1,
+ 0x9d412e0806e88aa5, 0x8e1f289560ee864e,
+ 0xc491798a08a2ad4e, 0xf1a6f2bab92a27e2,
+ 0xf5b5d7ec8acb58a2, 0xae10af696774b1db,
+ 0x9991a6f3d6bf1765, 0xacca6da1e0a8ef29,
+ 0xbff610b0cc6edd3f, 0x17fd090a58d32af3,
+ 0xeff394dcff8a948e, 0xddfc4b4cef07f5b0,
+ 0x95f83d0a1fb69cd9, 0x4abdaf101564f98e,
+ 0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f1,
+ 0xea53df5fd18d5513, 0x84c86189216dc5ed,
+ 0x92746b9be2f8552c, 0x32fd3cf5b4e49bb4,
+ 0xb7118682dbb66a77, 0x3fbc8c33221dc2a1,
+ 0xe4d5e82392a40515, 0xfabaf3feaa5334a,
+ 0x8f05b1163ba6832d, 0x29cb4d87f2a7400e,
+ 0xb2c71d5bca9023f8, 0x743e20e9ef511012,
+ 0xdf78e4b2bd342cf6, 0x914da9246b255416,
+ 0x8bab8eefb6409c1a, 0x1ad089b6c2f7548e,
+ 0xae9672aba3d0c320, 0xa184ac2473b529b1,
+ 0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741e,
+ 0x8865899617fb1871, 0x7e2fa67c7a658892,
+ 0xaa7eebfb9df9de8d, 0xddbb901b98feeab7,
+ 0xd51ea6fa85785631, 0x552a74227f3ea565,
+ 0x8533285c936b35de, 0xd53a88958f87275f,
+ 0xa67ff273b8460356, 0x8a892abaf368f137,
+ 0xd01fef10a657842c, 0x2d2b7569b0432d85,
+ 0x8213f56a67f6b29b, 0x9c3b29620e29fc73,
+ 0xa298f2c501f45f42, 0x8349f3ba91b47b8f,
+ 0xcb3f2f7642717713, 0x241c70a936219a73,
+ 0xfe0efb53d30dd4d7, 0xed238cd383aa0110,
+ 0x9ec95d1463e8a506, 0xf4363804324a40aa,
+ 0xc67bb4597ce2ce48, 0xb143c6053edcd0d5,
+ 0xf81aa16fdc1b81da, 0xdd94b7868e94050a,
+ 0x9b10a4e5e9913128, 0xca7cf2b4191c8326,
+ 0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f0,
+ 0xf24a01a73cf2dccf, 0xbc633b39673c8cec,
+ 0x976e41088617ca01, 0xd5be0503e085d813,
+ 0xbd49d14aa79dbc82, 0x4b2d8644d8a74e18,
+ 0xec9c459d51852ba2, 0xddf8e7d60ed1219e,
+ 0x93e1ab8252f33b45, 0xcabb90e5c942b503,
+ 0xb8da1662e7b00a17, 0x3d6a751f3b936243,
+ 0xe7109bfba19c0c9d, 0xcc512670a783ad4,
+ 0x906a617d450187e2, 0x27fb2b80668b24c5,
+ 0xb484f9dc9641e9da, 0xb1f9f660802dedf6,
+ 0xe1a63853bbd26451, 0x5e7873f8a0396973,
+ 0x8d07e33455637eb2, 0xdb0b487b6423e1e8,
+ 0xb049dc016abc5e5f, 0x91ce1a9a3d2cda62,
+ 0xdc5c5301c56b75f7, 0x7641a140cc7810fb,
+ 0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9d,
+ 0xac2820d9623bf429, 0x546345fa9fbdcd44,
+ 0xd732290fbacaf133, 0xa97c177947ad4095,
+ 0x867f59a9d4bed6c0, 0x49ed8eabcccc485d,
+ 0xa81f301449ee8c70, 0x5c68f256bfff5a74,
+ 0xd226fc195c6a2f8c, 0x73832eec6fff3111,
+ 0x83585d8fd9c25db7, 0xc831fd53c5ff7eab,
+ 0xa42e74f3d032f525, 0xba3e7ca8b77f5e55,
+ 0xcd3a1230c43fb26f, 0x28ce1bd2e55f35eb,
+ 0x80444b5e7aa7cf85, 0x7980d163cf5b81b3,
+ 0xa0555e361951c366, 0xd7e105bcc332621f,
+ 0xc86ab5c39fa63440, 0x8dd9472bf3fefaa7,
+ 0xfa856334878fc150, 0xb14f98f6f0feb951,
+ 0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d3,
+ 0xc3b8358109e84f07, 0xa862f80ec4700c8,
+ 0xf4a642e14c6262c8, 0xcd27bb612758c0fa,
+ 0x98e7e9cccfbd7dbd, 0x8038d51cb897789c,
+ 0xbf21e44003acdd2c, 0xe0470a63e6bd56c3,
+ 0xeeea5d5004981478, 0x1858ccfce06cac74,
+ 0x95527a5202df0ccb, 0xf37801e0c43ebc8,
+ 0xbaa718e68396cffd, 0xd30560258f54e6ba,
+ 0xe950df20247c83fd, 0x47c6b82ef32a2069,
+ 0x91d28b7416cdd27e, 0x4cdc331d57fa5441,
+ 0xb6472e511c81471d, 0xe0133fe4adf8e952,
+ 0xe3d8f9e563a198e5, 0x58180fddd97723a6,
+ 0x8e679c2f5e44ff8f, 0x570f09eaa7ea7648,
+ };
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <class unused>
+constexpr uint64_t
+ powers_template<unused>::power_of_five_128[number_of_entries];
+
+#endif
+
+using powers = powers_template<>;
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+#define SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+
+#include <cfloat>
+#include <cinttypes>
+#include <cmath>
+#include <cstdint>
+#include <cstdlib>
+#include <cstring>
+
+namespace simdjson_fast_float {
+
+// This will compute or rather approximate w * 5**q and return a pair of 64-bit
+// words approximating the result, with the "high" part corresponding to the
+// most significant bits and the low part corresponding to the least significant
+// bits.
+//
+template <int bit_precision>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+compute_product_approximation(int64_t q, uint64_t w) {
+ int const index = 2 * int(q - powers::smallest_power_of_five);
+ // For small values of q, e.g., q in [0,27], the answer is always exact
+ // because The line value128 firstproduct = full_multiplication(w,
+ // power_of_five_128[index]); gives the exact answer.
+ value128 firstproduct =
+ full_multiplication(w, powers::power_of_five_128[index]);
+ static_assert((bit_precision >= 0) && (bit_precision <= 64),
+ " precision should be in (0,64]");
+ constexpr uint64_t precision_mask =
+ (bit_precision < 64) ? (uint64_t(0xFFFFFFFFFFFFFFFF) >> bit_precision)
+ : uint64_t(0xFFFFFFFFFFFFFFFF);
+ if ((firstproduct.high & precision_mask) ==
+ precision_mask) { // could further guard with (lower + w < lower)
+ // regarding the second product, we only need secondproduct.high, but our
+ // expectation is that the compiler will optimize this extra work away if
+ // needed.
+ value128 secondproduct =
+ full_multiplication(w, powers::power_of_five_128[index + 1]);
firstproduct.low += secondproduct.high;
if (secondproduct.high > firstproduct.low) {
firstproduct.high++;
}
}
- uint64_t lower = firstproduct.low;
- uint64_t upper = firstproduct.high;
- uint64_t upperbit = upper >> 63;
- uint64_t mantissa = upper >> (upperbit + 9);
- lz += int(1 ^ upperbit);
- int64_t real_exponent = exponent - lz;
- if (real_exponent <= 0) {
- if (-real_exponent + 1 >= 64) {
- d = negative ? -0.0 : 0.0;
+ return firstproduct;
+}
+
+namespace detail {
+/**
+ * For q in (0,350), we have that
+ * f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ * floor(p) + q
+ * where
+ * p = log(5**q)/log(2) = q * log(5)/log(2)
+ *
+ * For negative values of q in (-400,0), we have that
+ * f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ * -ceil(p) + q
+ * where
+ * p = log(5**-q)/log(2) = -q * log(5)/log(2)
+ */
+constexpr simdjson_fastfloat_really_inline int32_t power(int32_t q) noexcept {
+ return (((152170 + 65536) * q) >> 16) + 63;
+}
+} // namespace detail
+
+// create an adjusted mantissa, biased by the invalid power2
+// for significant digits already multiplied by 10 ** q.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 adjusted_mantissa
+compute_error_scaled(int64_t q, uint64_t w, int lz) noexcept {
+ int hilz = int(w >> 63) ^ 1;
+ adjusted_mantissa answer;
+ answer.mantissa = w << hilz;
+ int bias = binary::mantissa_explicit_bits() - binary::minimum_exponent();
+ answer.power2 = int32_t(detail::power(int32_t(q)) + bias - hilz - lz - 62 +
+ invalid_am_bias);
+ return answer;
+}
+
+// w * 10 ** q, without rounding the representation up.
+// the power2 in the exponent will be adjusted by invalid_am_bias.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_error(int64_t q, uint64_t w) noexcept {
+ int lz = leading_zeroes(w);
+ w <<= lz;
+ value128 product =
+ compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+ return compute_error_scaled<binary>(q, product.high, lz);
+}
+
+// Computers w * 10 ** q.
+// The returned value should be a valid number that simply needs to be
+// packed. However, in some very rare cases, the computation will fail. In such
+// cases, we return an adjusted_mantissa with a negative power of 2: the caller
+// should recompute in such cases.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_float(int64_t q, uint64_t w) noexcept {
+ adjusted_mantissa answer;
+ if ((w == 0) || (q < binary::smallest_power_of_ten())) {
+ answer.power2 = 0;
+ answer.mantissa = 0;
+ // result should be zero
+ return answer;
+ }
+ if (q > binary::largest_power_of_ten()) {
+ // we want to get infinity:
+ answer.power2 = binary::infinite_power();
+ answer.mantissa = 0;
+ return answer;
+ }
+ // At this point in time q is in [powers::smallest_power_of_five,
+ // powers::largest_power_of_five].
+
+ // We want the most significant bit of i to be 1. Shift if needed.
+ int lz = leading_zeroes(w);
+ w <<= lz;
+
+ // The required precision is binary::mantissa_explicit_bits() + 3 because
+ // 1. We need the implicit bit
+ // 2. We need an extra bit for rounding purposes
+ // 3. We might lose a bit due to the "upperbit" routine (result too small,
+ // requiring a shift)
+
+ value128 product =
+ compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+ // The computed 'product' is always sufficient.
+ // Mathematical proof:
+ // Noble Mushtak and Daniel Lemire, Fast Number Parsing Without Fallback (to
+ // appear) See script/mushtak_lemire.py
+
+ // The "compute_product_approximation" function can be slightly slower than a
+ // branchless approach: value128 product = compute_product(q, w); but in
+ // practice, we can win big with the compute_product_approximation if its
+ // additional branch is easily predicted. Which is best is data specific.
+ int upperbit = int(product.high >> 63);
+ int shift = upperbit + 64 - binary::mantissa_explicit_bits() - 3;
+
+ answer.mantissa = product.high >> shift;
+
+ answer.power2 = int32_t(detail::power(int32_t(q)) + upperbit - lz -
+ binary::minimum_exponent());
+ if (answer.power2 <= 0) { // we have a subnormal?
+ // Here have that answer.power2 <= 0 so -answer.power2 >= 0
+ if (-answer.power2 + 1 >=
+ 64) { // if we have more than 64 bits below the minimum exponent, you
+ // have a zero for sure.
+ answer.power2 = 0;
+ answer.mantissa = 0;
+ // result should be zero
+ return answer;
+ }
+ // next line is safe because -answer.power2 + 1 < 64
+ answer.mantissa >>= -answer.power2 + 1;
+ // Thankfully, we can't have both "round-to-even" and subnormals because
+ // "round-to-even" only occurs for powers close to 0 in the 32-bit and
+ // and 64-bit case (with no more than 19 digits).
+ answer.mantissa += (answer.mantissa & 1); // round up
+ answer.mantissa >>= 1;
+ // There is a weird scenario where we don't have a subnormal but just.
+ // Suppose we start with 2.2250738585072013e-308, we end up
+ // with 0x3fffffffffffff x 2^-1023-53 which is technically subnormal
+ // whereas 0x40000000000000 x 2^-1023-53 is normal. Now, we need to round
+ // up 0x3fffffffffffff x 2^-1023-53 and once we do, we are no longer
+ // subnormal, but we can only know this after rounding.
+ // So we only declare a subnormal if we are smaller than the threshold.
+ answer.power2 =
+ (answer.mantissa < (uint64_t(1) << binary::mantissa_explicit_bits()))
+ ? 0
+ : 1;
+ return answer;
+ }
+
+ // usually, we round *up*, but if we fall right in between and and we have an
+ // even basis, we need to round down
+ // We are only concerned with the cases where 5**q fits in single 64-bit word.
+ if ((product.low <= 1) && (q >= binary::min_exponent_round_to_even()) &&
+ (q <= binary::max_exponent_round_to_even()) &&
+ ((answer.mantissa & 3) == 1)) { // we may fall between two floats!
+ // To be in-between two floats we need that in doing
+ // answer.mantissa = product.high >> (upperbit + 64 -
+ // binary::mantissa_explicit_bits() - 3);
+ // ... we dropped out only zeroes. But if this happened, then we can go
+ // back!!!
+ if ((answer.mantissa << shift) == product.high) {
+ answer.mantissa &= ~uint64_t(1); // flip it so that we do not round up
+ }
+ }
+
+ answer.mantissa += (answer.mantissa & 1); // round up
+ answer.mantissa >>= 1;
+ if (answer.mantissa >= (uint64_t(2) << binary::mantissa_explicit_bits())) {
+ answer.mantissa = (uint64_t(1) << binary::mantissa_explicit_bits());
+ answer.power2++; // undo previous addition
+ }
+
+ answer.mantissa &= ~(uint64_t(1) << binary::mantissa_explicit_bits());
+ if (answer.power2 >= binary::infinite_power()) { // infinity
+ answer.power2 = binary::infinite_power();
+ answer.mantissa = 0;
+ }
+ return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_BIGINT_H
+#define SIMDJSON_FASTFLOAT_BIGINT_H
+
+#include <algorithm>
+#include <cstdint>
+#include <climits>
+#include <cstring>
+
+
+namespace simdjson_fast_float {
+
+// the limb width: we want efficient multiplication of double the bits in
+// limb, or for 64-bit limbs, at least 64-bit multiplication where we can
+// extract the high and low parts efficiently. this is every 64-bit
+// architecture except for sparc, which emulates 128-bit multiplication.
+// we might have platforms where `CHAR_BIT` is not 8, so let's avoid
+// doing `8 * sizeof(limb)`.
+#if defined(SIMDJSON_FASTFLOAT_64BIT) && !defined(__sparc)
+#define SIMDJSON_FASTFLOAT_64BIT_LIMB 1
+typedef uint64_t limb;
+constexpr size_t limb_bits = 64;
+#else
+#define SIMDJSON_FASTFLOAT_32BIT_LIMB
+typedef uint32_t limb;
+constexpr size_t limb_bits = 32;
+#endif
+
+typedef span<limb> limb_span;
+
+// number of bits in a bigint. this needs to be at least the number
+// of bits required to store the largest bigint, which is
+// `log2(10**(digits + max_exp))`, or `log2(10**(767 + 342))`, or
+// ~3600 bits, so we round to 4000.
+constexpr size_t bigint_bits = 4000;
+constexpr size_t bigint_limbs = bigint_bits / limb_bits;
+
+// vector-like type that is allocated on the stack. the entire
+// buffer is pre-allocated, and only the length changes.
+template <uint16_t size> struct stackvec {
+ limb data[size];
+ // we never need more than 150 limbs
+ uint16_t length{0};
+
+ stackvec() = default;
+ stackvec(stackvec const &) = delete;
+ stackvec &operator=(stackvec const &) = delete;
+ stackvec(stackvec &&) = delete;
+ stackvec &operator=(stackvec &&other) = delete;
+
+ // create stack vector from existing limb span.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 stackvec(limb_span s) {
+ SIMDJSON_FASTFLOAT_ASSERT(try_extend(s));
+ }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 limb &operator[](size_t index) noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return data[index];
+ }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &operator[](size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ return data[index];
+ }
+
+ // index from the end of the container
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &rindex(size_t index) const noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+ size_t rindex = length - index - 1;
+ return data[rindex];
+ }
+
+ // set the length, without bounds checking.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void set_len(size_t len) noexcept {
+ length = uint16_t(len);
+ }
+
+ constexpr size_t len() const noexcept { return length; }
+
+ constexpr bool is_empty() const noexcept { return length == 0; }
+
+ constexpr size_t capacity() const noexcept { return size; }
+
+ // append item to vector, without bounds checking
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void push_unchecked(limb value) noexcept {
+ data[length] = value;
+ length++;
+ }
+
+ // append item to vector, returning if item was added
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 bool try_push(limb value) noexcept {
+ if (len() < capacity()) {
+ push_unchecked(value);
return true;
+ } else {
+ return false;
}
- mantissa >>= -real_exponent + 1;
- mantissa += (mantissa & 1);
- mantissa >>= 1;
- real_exponent = (mantissa < (uint64_t(1) << 52)) ? 0 : 1;
- d = to_double(mantissa, real_exponent, negative);
- return true;
}
- if ((lower <= 1) && (power >= -4) && (power <= 23) && ((mantissa & 3) == 1)) {
- if ((mantissa << (upperbit + 64 - 53 - 2)) == upper) {
- mantissa &= ~1;
+
+ // add items to the vector, from a span, without bounds checking
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 void extend_unchecked(limb_span s) noexcept {
+ limb *ptr = data + length;
+ std::copy_n(s.ptr, s.len(), ptr);
+ set_len(len() + s.len());
+ }
+
+ // try to add items to the vector, returning if items were added
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_extend(limb_span s) noexcept {
+ if (len() + s.len() <= capacity()) {
+ extend_unchecked(s);
+ return true;
+ } else {
+ return false;
}
}
- mantissa += mantissa & 1;
- mantissa >>= 1;
- if (mantissa >= (1ULL << 53)) {
- mantissa = (1ULL << 52);
- real_exponent++;
+
+ // resize the vector, without bounds checking
+ // if the new size is longer than the vector, assign value to each
+ // appended item.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20
+ void resize_unchecked(size_t new_len, limb value) noexcept {
+ if (new_len > len()) {
+ size_t count = new_len - len();
+ limb *first = data + len();
+ limb *last = first + count;
+ ::std::fill(first, last, value);
+ set_len(new_len);
+ } else {
+ set_len(new_len);
+ }
}
- mantissa &= ~(1ULL << 52);
- if (real_exponent > 2046) {
+
+ // try to resize the vector, returning if the vector was resized.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_resize(size_t new_len, limb value) noexcept {
+ if (new_len > capacity()) {
+ return false;
+ } else {
+ resize_unchecked(new_len, value);
+ return true;
+ }
+ }
+
+ // check if any limbs are non-zero after the given index.
+ // this needs to be done in reverse order, since the index
+ // is relative to the most significant limbs.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 bool nonzero(size_t index) const noexcept {
+ while (index < len()) {
+ if (rindex(index) != 0) {
+ return true;
+ }
+ index++;
+ }
return false;
}
- d = to_double(mantissa, real_exponent, negative);
+
+ // normalize the big integer, so most-significant zero limbs are removed.
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 void normalize() noexcept {
+ while (len() > 0 && rindex(0) == 0) {
+ length--;
+ }
+ }
+};
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+empty_hi64(bool &truncated) noexcept {
+ truncated = false;
+ return 0;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, bool &truncated) noexcept {
+ truncated = false;
+ int shl = leading_zeroes(r0);
+ return r0 << shl;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, uint64_t r1, bool &truncated) noexcept {
+ int shl = leading_zeroes(r0);
+ if (shl == 0) {
+ truncated = r1 != 0;
+ return r0;
+ } else {
+ int shr = 64 - shl;
+ truncated = (r1 << shl) != 0;
+ return (r0 << shl) | (r1 >> shr);
+ }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, bool &truncated) noexcept {
+ return uint64_hi64(r0, truncated);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, bool &truncated) noexcept {
+ uint64_t x0 = r0;
+ uint64_t x1 = r1;
+ return uint64_hi64((x0 << 32) | x1, truncated);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, uint32_t r2, bool &truncated) noexcept {
+ uint64_t x0 = r0;
+ uint64_t x1 = r1;
+ uint64_t x2 = r2;
+ return uint64_hi64(x0, (x1 << 32) | x2, truncated);
+}
+
+// add two small integers, checking for overflow.
+// we want an efficient operation. for msvc, where
+// we don't have built-in intrinsics, this is still
+// pretty fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_add(limb x, limb y, bool &overflow) noexcept {
+ limb z;
+// gcc and clang
+#if defined(__has_builtin)
+#if __has_builtin(__builtin_add_overflow)
+ if (!cpp20_and_in_constexpr()) {
+ overflow = __builtin_add_overflow(x, y, &z);
+ return z;
+ }
+#endif
+#endif
+
+ // generic, this still optimizes correctly on MSVC.
+ z = x + y;
+ overflow = z < x;
+ return z;
+}
+
+// multiply two small integers, getting both the high and low bits.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_mul(limb x, limb y, limb &carry) noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+#if defined(__SIZEOF_INT128__)
+ // GCC and clang both define it as an extension.
+ __uint128_t z = __uint128_t(x) * __uint128_t(y) + __uint128_t(carry);
+ carry = limb(z >> limb_bits);
+ return limb(z);
+#else
+ // fallback, no native 128-bit integer multiplication with carry.
+ // on msvc, this optimizes identically, somehow.
+ value128 z = full_multiplication(x, y);
+ bool overflow;
+ z.low = scalar_add(z.low, carry, overflow);
+ z.high += uint64_t(overflow); // cannot overflow
+ carry = z.high;
+ return z.low;
+#endif
+#else
+ uint64_t z = uint64_t(x) * uint64_t(y) + uint64_t(carry);
+ carry = limb(z >> limb_bits);
+ return limb(z);
+#endif
+}
+
+// add scalar value to bigint starting from offset.
+// used in grade school multiplication
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_add_from(stackvec<size> &vec, limb y,
+ size_t start) noexcept {
+ size_t index = start;
+ limb carry = y;
+ bool overflow;
+ while (carry != 0 && index < vec.len()) {
+ vec[index] = scalar_add(vec[index], carry, overflow);
+ carry = limb(overflow);
+ index += 1;
+ }
+ if (carry != 0) {
+ SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
+ }
return true;
}
-// Parses a single digit character and updates the integer value.
-consteval bool parse_digit(const char c, uint64_t &i) {
- const uint8_t digit = static_cast<uint8_t>(c - '0');
- if (digit > 9) {
- return false;
+// add scalar value to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+small_add(stackvec<size> &vec, limb y) noexcept {
+ return small_add_from(vec, y, 0);
+}
+
+// multiply bigint by scalar value.
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_mul(stackvec<size> &vec,
+ limb y) noexcept {
+ limb carry = 0;
+ for (size_t index = 0; index < vec.len(); index++) {
+ vec[index] = scalar_mul(vec[index], y, carry);
+ }
+ if (carry != 0) {
+ SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
}
- i = 10 * i + digit;
return true;
}
-// Parses a JSON float from a string starting at src.
-// Returns the parsed double and the number of characters consumed.
-consteval std::pair<double, size_t> parse_double(const char *src,
- const char *end) {
- auto get_value = [&](const char *pointer) -> char {
- if (pointer == end) {
- return '\0';
+// add bigint to bigint starting from index.
+// used in grade school multiplication
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_add_from(stackvec<size> &x, limb_span y,
+ size_t start) noexcept {
+ // the effective x buffer is from `xstart..x.len()`, so exit early
+ // if we can't get that current range.
+ if (x.len() < start || y.len() > x.len() - start) {
+ SIMDJSON_FASTFLOAT_TRY(x.try_resize(y.len() + start, 0));
+ }
+
+ bool carry = false;
+ for (size_t index = 0; index < y.len(); index++) {
+ limb xi = x[index + start];
+ limb yi = y[index];
+ bool c1 = false;
+ bool c2 = false;
+ xi = scalar_add(xi, yi, c1);
+ if (carry) {
+ xi = scalar_add(xi, 1, c2);
}
- return *pointer;
+ x[index + start] = xi;
+ carry = c1 | c2;
+ }
+
+ // handle overflow
+ if (carry) {
+ SIMDJSON_FASTFLOAT_TRY(small_add_from(x, 1, y.len() + start));
+ }
+ return true;
+}
+
+// add bigint to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+large_add_from(stackvec<size> &x, limb_span y) noexcept {
+ return large_add_from(x, y, 0);
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool long_mul(stackvec<size> &x, limb_span y) noexcept {
+ limb_span xs = limb_span(x.data, x.len());
+ stackvec<size> z(xs);
+ limb_span zs = limb_span(z.data, z.len());
+
+ if (y.len() != 0) {
+ limb y0 = y[0];
+ SIMDJSON_FASTFLOAT_TRY(small_mul(x, y0));
+ for (size_t index = 1; index < y.len(); index++) {
+ limb yi = y[index];
+ stackvec<size> zi;
+ if (yi != 0) {
+ // re-use the same buffer throughout
+ zi.set_len(0);
+ SIMDJSON_FASTFLOAT_TRY(zi.try_extend(zs));
+ SIMDJSON_FASTFLOAT_TRY(small_mul(zi, yi));
+ limb_span zis = limb_span(zi.data, zi.len());
+ SIMDJSON_FASTFLOAT_TRY(large_add_from(x, zis, index));
+ }
+ }
+ }
+
+ x.normalize();
+ return true;
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_mul(stackvec<size> &x, limb_span y) noexcept {
+ if (y.len() == 1) {
+ SIMDJSON_FASTFLOAT_TRY(small_mul(x, y[0]));
+ } else {
+ SIMDJSON_FASTFLOAT_TRY(long_mul(x, y));
+ }
+ return true;
+}
+
+template <typename = void> struct pow5_tables {
+ static constexpr uint32_t large_step = 135;
+ static constexpr uint64_t small_power_of_5[] = {
+ 1UL,
+ 5UL,
+ 25UL,
+ 125UL,
+ 625UL,
+ 3125UL,
+ 15625UL,
+ 78125UL,
+ 390625UL,
+ 1953125UL,
+ 9765625UL,
+ 48828125UL,
+ 244140625UL,
+ 1220703125UL,
+ 6103515625UL,
+ 30517578125UL,
+ 152587890625UL,
+ 762939453125UL,
+ 3814697265625UL,
+ 19073486328125UL,
+ 95367431640625UL,
+ 476837158203125UL,
+ 2384185791015625UL,
+ 11920928955078125UL,
+ 59604644775390625UL,
+ 298023223876953125UL,
+ 1490116119384765625UL,
+ 7450580596923828125UL,
};
- const char *srcinit = src;
- bool negative = (get_value(src) == '-');
- src += uint8_t(negative);
- uint64_t i = 0;
- const char *p = src;
- p += parse_digit(get_value(p), i);
- bool leading_zero = (i == 0);
- while (parse_digit(get_value(p), i)) {
- p++;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ constexpr static limb large_power_of_5[] = {
+ 1414648277510068013UL, 9180637584431281687UL, 4539964771860779200UL,
+ 10482974169319127550UL, 198276706040285095UL};
+#else
+ constexpr static limb large_power_of_5[] = {
+ 4279965485U, 329373468U, 4020270615U, 2137533757U, 4287402176U,
+ 1057042919U, 1071430142U, 2440757623U, 381945767U, 46164893U};
+#endif
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint32_t pow5_tables<T>::large_step;
+
+template <typename T> constexpr uint64_t pow5_tables<T>::small_power_of_5[];
+
+template <typename T> constexpr limb pow5_tables<T>::large_power_of_5[];
+
+#endif
+
+// big integer type. implements a small subset of big integer
+// arithmetic, using simple algorithms since asymptotically
+// faster algorithms are slower for a small number of limbs.
+// all operations assume the big-integer is normalized.
+struct bigint : pow5_tables<> {
+ // storage of the limbs, in little-endian order.
+ stackvec<bigint_limbs> vec;
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint() : vec() {}
+
+ bigint(bigint const &) = delete;
+ bigint &operator=(bigint const &) = delete;
+ bigint(bigint &&) = delete;
+ bigint &operator=(bigint &&other) = delete;
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint(uint64_t value) : vec() {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ vec.push_unchecked(value);
+#else
+ vec.push_unchecked(uint32_t(value));
+ vec.push_unchecked(uint32_t(value >> 32));
+#endif
+ vec.normalize();
}
- if (p == src) {
- simdjson_consteval_error("Invalid float value");
+
+ // get the high 64 bits from the vector, and if bits were truncated.
+ // this is to get the significant digits for the float.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t hi64(bool &truncated) const noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ if (vec.len() == 0) {
+ return empty_hi64(truncated);
+ } else if (vec.len() == 1) {
+ return uint64_hi64(vec.rindex(0), truncated);
+ } else {
+ uint64_t result = uint64_hi64(vec.rindex(0), vec.rindex(1), truncated);
+ truncated |= vec.nonzero(2);
+ return result;
+ }
+#else
+ if (vec.len() == 0) {
+ return empty_hi64(truncated);
+ } else if (vec.len() == 1) {
+ return uint32_hi64(vec.rindex(0), truncated);
+ } else if (vec.len() == 2) {
+ return uint32_hi64(vec.rindex(0), vec.rindex(1), truncated);
+ } else {
+ uint64_t result =
+ uint32_hi64(vec.rindex(0), vec.rindex(1), vec.rindex(2), truncated);
+ truncated |= vec.nonzero(3);
+ return result;
+ }
+#endif
}
- if ((leading_zero && p != src + 1)) {
- simdjson_consteval_error("Invalid float value");
+
+ // compare two big integers, returning the large value.
+ // assumes both are normalized. if the return value is
+ // negative, other is larger, if the return value is
+ // positive, this is larger, otherwise they are equal.
+ // the limbs are stored in little-endian order, so we
+ // must compare the limbs in ever order.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int compare(bigint const &other) const noexcept {
+ if (vec.len() > other.vec.len()) {
+ return 1;
+ } else if (vec.len() < other.vec.len()) {
+ return -1;
+ } else {
+ for (size_t index = vec.len(); index > 0; index--) {
+ limb xi = vec[index - 1];
+ limb yi = other.vec[index - 1];
+ if (xi > yi) {
+ return 1;
+ } else if (xi < yi) {
+ return -1;
+ }
+ }
+ return 0;
+ }
}
- int64_t exponent = 0;
- bool overflow;
- if (get_value(p) == '.') {
- p++;
- const char *start_decimal_digits = p;
- if (!parse_digit(get_value(p), i)) {
- simdjson_consteval_error("Invalid float value");
- } // no decimal digits
- p++;
- while (parse_digit(get_value(p), i)) {
- p++;
+
+ // shift left each limb n bits, carrying over to the new limb
+ // returns true if we were able to shift all the digits.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_bits(size_t n) noexcept {
+ // Internally, for each item, we shift left by n, and add the previous
+ // right shifted limb-bits.
+ // For example, we transform (for u8) shifted left 2, to:
+ // b10100100 b01000010
+ // b10 b10010001 b00001000
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n < sizeof(limb) * 8);
+
+ size_t shl = n;
+ size_t shr = limb_bits - shl;
+ limb prev = 0;
+ for (size_t index = 0; index < vec.len(); index++) {
+ limb xi = vec[index];
+ vec[index] = (xi << shl) | (prev >> shr);
+ prev = xi;
}
- exponent = -(p - start_decimal_digits);
- overflow = p - src - 1 > 19;
- if (overflow && leading_zero) {
- const char *start_digits = src + 2;
- while (get_value(start_digits) == '0') {
- start_digits++;
+
+ limb carry = prev >> shr;
+ if (carry != 0) {
+ return vec.try_push(carry);
+ }
+ return true;
+ }
+
+ // move the limbs left by `n` limbs.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_limbs(size_t n) noexcept {
+ SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+ if (n + vec.len() > vec.capacity()) {
+ return false;
+ } else if (!vec.is_empty()) {
+ // move limbs
+ limb *dst = vec.data + n;
+ limb const *src = vec.data;
+ std::copy_backward(src, src + vec.len(), dst + vec.len());
+ // fill in empty limbs
+ limb *first = vec.data;
+ limb *last = first + n;
+ ::std::fill(first, last, 0);
+ vec.set_len(n + vec.len());
+ return true;
+ } else {
+ return true;
+ }
+ }
+
+ // move the limbs left by `n` bits.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl(size_t n) noexcept {
+ size_t rem = n % limb_bits;
+ size_t div = n / limb_bits;
+ if (rem != 0) {
+ SIMDJSON_FASTFLOAT_TRY(shl_bits(rem));
+ }
+ if (div != 0) {
+ SIMDJSON_FASTFLOAT_TRY(shl_limbs(div));
+ }
+ return true;
+ }
+
+ // get the number of leading zeros in the bigint.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int ctlz() const noexcept {
+ if (vec.is_empty()) {
+ return 0;
+ } else {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ return leading_zeroes(vec.rindex(0));
+#else
+ // no use defining a specialized leading_zeroes for a 32-bit type.
+ uint64_t r0 = vec.rindex(0);
+ return leading_zeroes(r0 << 32);
+#endif
+ }
+ }
+
+ // get the number of bits in the bigint.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 int bit_length() const noexcept {
+ int lz = ctlz();
+ return int(limb_bits * vec.len()) - lz;
+ }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool mul(limb y) noexcept { return small_mul(vec, y); }
+
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool add(limb y) noexcept { return small_add(vec, y); }
+
+ // multiply as if by 2 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow2(uint32_t exp) noexcept { return shl(exp); }
+
+ // multiply as if by 5 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow5(uint32_t exp) noexcept {
+ // multiply by a power of 5
+ size_t large_length = sizeof(large_power_of_5) / sizeof(limb);
+ limb_span large = limb_span(large_power_of_5, large_length);
+ while (exp >= large_step) {
+ SIMDJSON_FASTFLOAT_TRY(large_mul(vec, large));
+ exp -= large_step;
+ }
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ uint32_t small_step = 27;
+ limb max_native = 7450580596923828125UL;
+#else
+ uint32_t small_step = 13;
+ limb max_native = 1220703125U;
+#endif
+ while (exp >= small_step) {
+ SIMDJSON_FASTFLOAT_TRY(small_mul(vec, max_native));
+ exp -= small_step;
+ }
+ if (exp != 0) {
+ // Work around clang bug https://godbolt.org/z/zedh7rrhc
+ // This is similar to https://github.com/llvm/llvm-project/issues/47746,
+ // except the workaround described there don't work here
+ SIMDJSON_FASTFLOAT_TRY(small_mul(vec, limb((static_cast<void>(small_power_of_5[0]),
+ small_power_of_5[exp]))));
+ }
+
+ return true;
+ }
+
+ // multiply as if by 10 raised to a power.
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow10(uint32_t exp) noexcept {
+ SIMDJSON_FASTFLOAT_TRY(pow5(exp));
+ return pow2(exp);
+ }
+};
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+#define SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+
+
+namespace simdjson_fast_float {
+
+// 1e0 to 1e19
+constexpr static uint64_t powers_of_ten_uint64[] = {1UL,
+ 10UL,
+ 100UL,
+ 1000UL,
+ 10000UL,
+ 100000UL,
+ 1000000UL,
+ 10000000UL,
+ 100000000UL,
+ 1000000000UL,
+ 10000000000UL,
+ 100000000000UL,
+ 1000000000000UL,
+ 10000000000000UL,
+ 100000000000000UL,
+ 1000000000000000UL,
+ 10000000000000000UL,
+ 100000000000000000UL,
+ 1000000000000000000UL,
+ 10000000000000000000UL};
+
+// calculate the exponent, in scientific notation, of the number.
+// this algorithm is not even close to optimized, but it has no practical
+// effect on performance: in order to have a faster algorithm, we'd need
+// to slow down performance for faster algorithms, and this is still fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int32_t
+scientific_exponent(uint64_t mantissa, int32_t exponent) noexcept {
+ while (mantissa >= 10000) {
+ mantissa /= 10000;
+ exponent += 4;
+ }
+ while (mantissa >= 100) {
+ mantissa /= 100;
+ exponent += 2;
+ }
+ while (mantissa >= 10) {
+ mantissa /= 10;
+ exponent += 1;
+ }
+ return exponent;
+}
+
+// this converts a native floating-point number to an extended-precision float.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended(T value) noexcept {
+ using equiv_uint = equiv_uint_t<T>;
+ constexpr equiv_uint exponent_mask = binary_format<T>::exponent_mask();
+ constexpr equiv_uint mantissa_mask = binary_format<T>::mantissa_mask();
+ constexpr equiv_uint hidden_bit_mask = binary_format<T>::hidden_bit_mask();
+
+ adjusted_mantissa am;
+ int32_t bias = binary_format<T>::mantissa_explicit_bits() -
+ binary_format<T>::minimum_exponent();
+ equiv_uint bits;
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+ bits = std::bit_cast<equiv_uint>(value);
+#else
+ ::memcpy(&bits, &value, sizeof(T));
+#endif
+ if ((bits & exponent_mask) == 0) {
+ // denormal
+ am.power2 = 1 - bias;
+ am.mantissa = bits & mantissa_mask;
+ } else {
+ // normal
+ am.power2 = int32_t((bits & exponent_mask) >>
+ binary_format<T>::mantissa_explicit_bits());
+ am.power2 -= bias;
+ am.mantissa = (bits & mantissa_mask) | hidden_bit_mask;
+ }
+
+ return am;
+}
+
+// get the extended precision value of the halfway point between b and b+u.
+// we are given a native float that represents b, so we need to adjust it
+// halfway between b and b+u.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended_halfway(T value) noexcept {
+ adjusted_mantissa am = to_extended(value);
+ am.mantissa <<= 1;
+ am.mantissa += 1;
+ am.power2 -= 1;
+ return am;
+}
+
+// round an extended-precision float to the nearest machine float.
+template <typename T, typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void round(adjusted_mantissa &am,
+ callback cb) noexcept {
+ int32_t mantissa_shift = 64 - binary_format<T>::mantissa_explicit_bits() - 1;
+ if (-am.power2 >= mantissa_shift) {
+ // have a denormal float
+ int32_t shift = -am.power2 + 1;
+ cb(am, (shift < 64 ? shift : 64));
+ // check for round-up: if rounding-nearest carried us to the hidden bit.
+ am.power2 = (am.mantissa <
+ (uint64_t(1) << binary_format<T>::mantissa_explicit_bits()))
+ ? 0
+ : 1;
+ return;
+ }
+
+ // have a normal float, use the default shift.
+ cb(am, mantissa_shift);
+
+ // check for carry
+ if (am.mantissa >=
+ (uint64_t(2) << binary_format<T>::mantissa_explicit_bits())) {
+ am.mantissa = (uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+ am.power2++;
+ }
+
+ // check for infinite: we could have carried to an infinite power
+ am.mantissa &= ~(uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+ if (am.power2 >= binary_format<T>::infinite_power()) {
+ am.power2 = binary_format<T>::infinite_power();
+ am.mantissa = 0;
+ }
+}
+
+template <typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_nearest_tie_even(adjusted_mantissa &am, int32_t shift,
+ callback cb) noexcept {
+ uint64_t const mask = (shift == 64) ? UINT64_MAX : (uint64_t(1) << shift) - 1;
+ uint64_t const halfway = (shift == 0) ? 0 : uint64_t(1) << (shift - 1);
+ uint64_t truncated_bits = am.mantissa & mask;
+ bool is_above = truncated_bits > halfway;
+ bool is_halfway = truncated_bits == halfway;
+
+ // shift digits into position
+ if (shift == 64) {
+ am.mantissa = 0;
+ } else {
+ am.mantissa >>= shift;
+ }
+ am.power2 += shift;
+
+ bool is_odd = (am.mantissa & 1) == 1;
+ am.mantissa += uint64_t(cb(is_odd, is_halfway, is_above));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_down(adjusted_mantissa &am, int32_t shift) noexcept {
+ if (shift == 64) {
+ am.mantissa = 0;
+ } else {
+ am.mantissa >>= shift;
+ }
+ am.power2 += shift;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+skip_zeros(UC const *&first, UC const *last) noexcept {
+ uint64_t val;
+ while (!cpp20_and_in_constexpr() &&
+ std::distance(first, last) >= int_cmp_len<UC>()) {
+ ::memcpy(&val, first, sizeof(uint64_t));
+ if (val != int_cmp_zeros<UC>()) {
+ break;
+ }
+ first += int_cmp_len<UC>();
+ }
+ while (first != last) {
+ if (*first != UC('0')) {
+ break;
+ }
+ first++;
+ }
+}
+
+// determine if any non-zero digits were truncated.
+// all characters must be valid digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(UC const *first, UC const *last) noexcept {
+ // do 8-bit optimizations, can just compare to 8 literal 0s.
+ uint64_t val;
+ while (!cpp20_and_in_constexpr() &&
+ std::distance(first, last) >= int_cmp_len<UC>()) {
+ ::memcpy(&val, first, sizeof(uint64_t));
+ if (val != int_cmp_zeros<UC>()) {
+ return true;
+ }
+ first += int_cmp_len<UC>();
+ }
+ while (first != last) {
+ if (*first != UC('0')) {
+ return true;
+ }
+ ++first;
+ }
+ return false;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(span<UC const> s) noexcept {
+ return is_truncated(s.ptr, s.ptr + s.len());
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_eight_digits(UC const *&p, limb &value, size_t &counter,
+ size_t &count) noexcept {
+ value = value * 100000000 + parse_eight_digits_unrolled(p);
+ p += 8;
+ counter += 8;
+ count += 8;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+parse_one_digit(UC const *&p, limb &value, size_t &counter,
+ size_t &count) noexcept {
+ value = value * 10 + limb(*p - UC('0'));
+ p++;
+ counter++;
+ count++;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+add_native(bigint &big, limb power, limb value) noexcept {
+ big.mul(power);
+ big.add(value);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+round_up_bigint(bigint &big, size_t &count) noexcept {
+ // need to round-up the digits, but need to avoid rounding
+ // ....9999 to ...10000, which could cause a false halfway point.
+ add_native(big, 10, 1);
+ count++;
+}
+
+// parse the significant digits into a big integer
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_mantissa(bigint &result, parsed_number_string_t<UC> &num,
+ size_t max_digits, size_t &digits) noexcept {
+ // try to minimize the number of big integer and scalar multiplication.
+ // therefore, try to parse 8 digits at a time, and multiply by the largest
+ // scalar value (9 or 19 digits) for each step.
+ size_t counter = 0;
+ digits = 0;
+ limb value = 0;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+ size_t step = 19;
+#else
+ size_t step = 9;
+#endif
+
+ // process all integer digits.
+ UC const *p = num.integer.ptr;
+ UC const *pend = p + num.integer.len();
+ skip_zeros(p, pend);
+ // process all digits, in increments of step per loop
+ while (p != pend) {
+ while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+ (max_digits - digits >= 8)) {
+ parse_eight_digits(p, value, counter, digits);
+ }
+ while (counter < step && p != pend && digits < max_digits) {
+ parse_one_digit(p, value, counter, digits);
+ }
+ if (digits == max_digits) {
+ // add the temporary value, then check if we've truncated any digits
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ bool truncated = is_truncated(p, pend);
+ if (num.fraction.ptr != nullptr) {
+ truncated |= is_truncated(num.fraction);
+ }
+ if (truncated) {
+ round_up_bigint(result, digits);
+ }
+ return;
+ } else {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ counter = 0;
+ value = 0;
+ }
+ }
+
+ // add our fraction digits, if they're available.
+ if (num.fraction.ptr != nullptr) {
+ p = num.fraction.ptr;
+ pend = p + num.fraction.len();
+ if (digits == 0) {
+ skip_zeros(p, pend);
+ }
+ // process all digits, in increments of step per loop
+ while (p != pend) {
+ while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+ (max_digits - digits >= 8)) {
+ parse_eight_digits(p, value, counter, digits);
+ }
+ while (counter < step && p != pend && digits < max_digits) {
+ parse_one_digit(p, value, counter, digits);
+ }
+ if (digits == max_digits) {
+ // add the temporary value, then check if we've truncated any digits
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ bool truncated = is_truncated(p, pend);
+ if (truncated) {
+ round_up_bigint(result, digits);
+ }
+ return;
+ } else {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ counter = 0;
+ value = 0;
}
- overflow = p - start_digits > 19;
}
+ }
+
+ if (counter != 0) {
+ add_native(result, limb(powers_of_ten_uint64[counter]), value);
+ }
+}
+
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+positive_digit_comp(bigint &bigmant, int32_t exponent) noexcept {
+ SIMDJSON_FASTFLOAT_ASSERT(bigmant.pow10(uint32_t(exponent)));
+ adjusted_mantissa answer;
+ bool truncated;
+ answer.mantissa = bigmant.hi64(truncated);
+ int bias = binary_format<T>::mantissa_explicit_bits() -
+ binary_format<T>::minimum_exponent();
+ answer.power2 = bigmant.bit_length() - 64 + bias;
+
+ round<T>(answer, [truncated](adjusted_mantissa &a, int32_t shift) {
+ round_nearest_tie_even(
+ a, shift,
+ [truncated](bool is_odd, bool is_halfway, bool is_above) -> bool {
+ return is_above || (is_halfway && truncated) ||
+ (is_odd && is_halfway);
+ });
+ });
+
+ return answer;
+}
+
+// the scaling here is quite simple: we have, for the real digits `m * 10^e`,
+// and for the theoretical digits `n * 2^f`. Since `e` is always negative,
+// to scale them identically, we do `n * 2^f * 5^-f`, so we now have `m * 2^e`.
+// we then need to scale by `2^(f- e)`, and then the two significant digits
+// are of the same magnitude.
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa negative_digit_comp(
+ bigint &bigmant, adjusted_mantissa am, int32_t exponent) noexcept {
+ bigint &real_digits = bigmant;
+ int32_t real_exp = exponent;
+
+ // get the value of `b`, rounded down, and get a bigint representation of b+h
+ adjusted_mantissa am_b = am;
+ // gcc7 buf: use a lambda to remove the noexcept qualifier bug with
+ // -Wnoexcept-type.
+ round<T>(am_b,
+ [](adjusted_mantissa &a, int32_t shift) { round_down(a, shift); });
+ T b;
+ to_float(false, am_b, b);
+ adjusted_mantissa theor = to_extended_halfway(b);
+ bigint theor_digits(theor.mantissa);
+ int32_t theor_exp = theor.power2;
+
+ // scale real digits and theor digits to be same power.
+ int32_t pow2_exp = theor_exp - real_exp;
+ uint32_t pow5_exp = uint32_t(-real_exp);
+ if (pow5_exp != 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow5(pow5_exp));
+ }
+ if (pow2_exp > 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow2(uint32_t(pow2_exp)));
+ } else if (pow2_exp < 0) {
+ SIMDJSON_FASTFLOAT_ASSERT(real_digits.pow2(uint32_t(-pow2_exp)));
+ }
+
+ // compare digits, and use it to direct rounding
+ int ord = real_digits.compare(theor_digits);
+ adjusted_mantissa answer = am;
+ round<T>(answer, [ord](adjusted_mantissa &a, int32_t shift) {
+ round_nearest_tie_even(
+ a, shift, [ord](bool is_odd, bool _, bool __) -> bool {
+ static_cast<void>(_); // not needed, since we've done our comparison
+ static_cast<void>(__); // not needed, since we've done our comparison
+ if (ord > 0) {
+ return true;
+ } else if (ord < 0) {
+ return false;
+ } else {
+ return is_odd;
+ }
+ });
+ });
+
+ return answer;
+}
+
+// parse the significant digits as a big integer to unambiguously round
+// the significant digits. here, we are trying to determine how to round
+// an extended float representation close to `b+h`, halfway between `b`
+// (the float rounded-down) and `b+u`, the next positive float. this
+// algorithm is always correct, and uses one of two approaches. when
+// the exponent is positive relative to the significant digits (such as
+// 1234), we create a big-integer representation, get the high 64-bits,
+// determine if any lower bits are truncated, and use that to direct
+// rounding. in case of a negative exponent relative to the significant
+// digits (such as 1.2345), we create a theoretical representation of
+// `b` as a big-integer type, scaled to the same binary exponent as
+// the actual digits. we then compare the big integer representations
+// of both, and use that to direct rounding.
+template <typename T, typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+digit_comp(parsed_number_string_t<UC> &num, adjusted_mantissa am) noexcept {
+ // remove the invalid exponent bias
+ am.power2 -= invalid_am_bias;
+
+ int32_t sci_exp =
+ scientific_exponent(num.mantissa, static_cast<int32_t>(num.exponent));
+ size_t max_digits = binary_format<T>::max_digits();
+ size_t digits = 0;
+ bigint bigmant;
+ parse_mantissa(bigmant, num, max_digits, digits);
+ // can't underflow, since digits is at most max_digits.
+ int32_t exponent = sci_exp + 1 - int32_t(digits);
+ if (exponent >= 0) {
+ return positive_digit_comp<T>(bigmant, exponent);
} else {
- overflow = p - src > 19;
+ return negative_digit_comp<T>(bigmant, am, exponent);
+ }
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+#define SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+
+
+#include <cmath>
+#include <cstring>
+#include <limits>
+#include <system_error>
+
+namespace simdjson_fast_float {
+
+namespace detail {
+/**
+ * Special case +inf, -inf, nan, infinity, -infinity.
+ * The case comparisons could be made much faster given that we know that the
+ * strings a null-free and fixed.
+ **/
+template <typename T, typename UC>
+from_chars_result_t<UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR14 parse_infnan(UC const *first, UC const *last,
+ T &value, chars_format fmt) noexcept {
+ from_chars_result_t<UC> answer{};
+ answer.ptr = first;
+ answer.ec = std::errc(); // be optimistic
+ // assume first < last, so dereference without checks;
+ bool const minusSign = (*first == UC('-'));
+ // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+ if ((*first == UC('-')) ||
+ (uint64_t(fmt & chars_format::allow_leading_plus) &&
+ (*first == UC('+')))) {
+ ++first;
+ }
+ if (last - first >= 3) {
+ if (simdjson_fastfloat_strncasecmp3(first, str_const_nan<UC>())) {
+ answer.ptr = (first += 3);
+ value = minusSign ? -std::numeric_limits<T>::quiet_NaN()
+ : std::numeric_limits<T>::quiet_NaN();
+ // Check for possible nan(n-char-seq-opt), C++17 20.19.3.7,
+ // C11 7.20.1.3.3. At least MSVC produces nan(ind) and nan(snan).
+ if (first != last && *first == UC('(')) {
+ for (UC const *ptr = first + 1; ptr != last; ++ptr) {
+ if (*ptr == UC(')')) {
+ answer.ptr = ptr + 1; // valid nan(n-char-seq-opt)
+ break;
+ } else if (!((UC('a') <= *ptr && *ptr <= UC('z')) ||
+ (UC('A') <= *ptr && *ptr <= UC('Z')) ||
+ (UC('0') <= *ptr && *ptr <= UC('9')) || *ptr == UC('_')))
+ break; // forbidden char, not nan(n-char-seq-opt)
+ }
+ }
+ return answer;
+ }
+ if (simdjson_fastfloat_strncasecmp3(first, str_const_inf<UC>())) {
+ if ((last - first >= 8) &&
+ simdjson_fastfloat_strncasecmp5(first + 3, str_const_inf<UC>() + 3)) {
+ answer.ptr = first + 8;
+ } else {
+ answer.ptr = first + 3;
+ }
+ value = minusSign ? -std::numeric_limits<T>::infinity()
+ : std::numeric_limits<T>::infinity();
+ return answer;
+ }
}
- if (overflow) {
- simdjson_consteval_error(
- "Overflow while computing the float value: too many digits");
+ answer.ec = std::errc::invalid_argument;
+ return answer;
+}
+
+/**
+ * Returns true if the floating-pointing rounding mode is to 'nearest'.
+ * It is the default on most system. This function is meant to be inexpensive.
+ * Credit : @mwalcott3
+ */
+simdjson_fastfloat_really_inline bool rounds_to_nearest() noexcept {
+ // https://lemire.me/blog/2020/06/26/gcc-not-nearest/
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+ return false;
+#endif
+ // See
+ // A fast function to check your floating-point rounding mode
+ // https://lemire.me/blog/2022/11/16/a-fast-function-to-check-your-floating-point-rounding-mode/
+ //
+ // This function is meant to be equivalent to :
+ // prior: #include <cfenv>
+ // return fegetround() == FE_TONEAREST;
+ // However, it is expected to be much faster than the fegetround()
+ // function call.
+ //
+ // The volatile keyword prevents the compiler from computing the function
+ // at compile-time.
+ // There might be other ways to prevent compile-time optimizations (e.g.,
+ // asm). The value does not need to be std::numeric_limits<float>::min(), any
+ // small value so that 1 + x should round to 1 would do (after accounting for
+ // excess precision, as in 387 instructions).
+ static float volatile fmin = (std::numeric_limits<float>::min)();
+ float fmini = fmin; // we copy it so that it gets loaded at most once.
+//
+// Explanation:
+// Only when fegetround() == FE_TONEAREST do we have that
+// fmin + 1.0f == 1.0f - fmin.
+//
+// FE_UPWARD:
+// fmin + 1.0f > 1
+// 1.0f - fmin == 1
+//
+// FE_DOWNWARD or FE_TOWARDZERO:
+// fmin + 1.0f == 1
+// 1.0f - fmin < 1
+//
+// Note: This may fail to be accurate if fast-math has been
+// enabled, as rounding conventions may not apply.
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+// todo: is there a VS warning?
+// see
+// https://stackoverflow.com/questions/46079446/is-there-a-warning-for-floating-point-equality-checking-in-visual-studio-2013
+#elif defined(__clang__)
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Wfloat-equal"
+#elif defined(__GNUC__)
+#pragma GCC diagnostic push
+#pragma GCC diagnostic ignored "-Wfloat-equal"
+#endif
+ return (fmini + 1.0f == 1.0f - fmini);
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#elif defined(__clang__)
+#pragma clang diagnostic pop
+#elif defined(__GNUC__)
+#pragma GCC diagnostic pop
+#endif
+}
+
+} // namespace detail
+
+template <typename T> struct from_chars_caller {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_advanced(first, last, value, options);
}
- if (get_value(p) == 'e' || get_value(p) == 'E') {
- p++;
- bool exp_neg = get_value(p) == '-';
- p += exp_neg || get_value(p) == '+';
- uint64_t exp = 0;
- const char *start_exp_digits = p;
- while (parse_digit(get_value(p), exp)) {
- p++;
+};
+
+#ifdef __STDCPP_FLOAT32_T__
+template <> struct from_chars_caller<std::float32_t> {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, std::float32_t &value,
+ parse_options_t<UC> options) noexcept {
+ // if std::float32_t is defined, and we are in C++23 mode; macro set for
+ // float32; set value to float due to equivalence between float and
+ // float32_t
+ float val = 0.0f;
+ auto ret = from_chars_advanced(first, last, val, options);
+ value = val;
+ return ret;
+ }
+};
+#endif
+
+#ifdef __STDCPP_FLOAT64_T__
+template <> struct from_chars_caller<std::float64_t> {
+ template <typename UC>
+ SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, std::float64_t &value,
+ parse_options_t<UC> options) noexcept {
+ // if std::float64_t is defined, and we are in C++23 mode; macro set for
+ // float64; set value as double due to equivalence between double and
+ // float64_t
+ double val = 0.0;
+ auto ret = from_chars_advanced(first, last, val, options);
+ value = val;
+ return ret;
+ }
+};
+#endif
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+ chars_format fmt /*= chars_format::general*/) noexcept {
+ return from_chars_caller<T>::call(first, last, value,
+ parse_options_t<UC>(fmt));
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+clinger_fast_path_impl(uint64_t mantissa, int64_t exponent, bool is_negative,
+ T &value) noexcept {
+ // The implementation of the Clinger's fast path is convoluted because
+ // we want round-to-nearest in all cases, irrespective of the rounding mode
+ // selected on the thread.
+ // We proceed optimistically, assuming that detail::rounds_to_nearest()
+ // returns true.
+ if (binary_format<T>::min_exponent_fast_path() <= exponent &&
+ exponent <= binary_format<T>::max_exponent_fast_path() &&
+ mantissa <= binary_format<T>::max_mantissa_fast_path()) {
+ // The mantissa bound above is a necessary condition for BOTH branches
+ // below: the rounding-mode-dependent branch checks the tighter
+ // max_mantissa_fast_path(exponent) <= max_mantissa_fast_path(). Testing
+ // it before detail::rounds_to_nearest() spares long-mantissa inputs
+ // (which can never take the fast path) the volatile-float probe.
+ //
+ // Unfortunately, the conventional Clinger's fast path is only possible
+ // when the system rounds to the nearest float.
+ //
+ // We expect the next branch to almost always be selected.
+ // We could check it first (before the previous branch), but
+ // there might be performance advantages at having the check
+ // be last.
+ if (!cpp20_and_in_constexpr() && detail::rounds_to_nearest()) {
+ // We have that fegetround() == FE_TONEAREST.
+ // Next is Clinger's fast path.
+ value = T(mantissa);
+ if (exponent < 0) {
+ value = value / binary_format<T>::exact_power_of_ten(-exponent);
+ } else {
+ value = value * binary_format<T>::exact_power_of_ten(exponent);
+ }
+ if (is_negative) {
+ value = -value;
+ }
+ return true;
+ } else {
+ // We do not have that fegetround() == FE_TONEAREST.
+ // Next is a modified Clinger's fast path, inspired by Jakub Jelinek's
+ // proposal
+ if (exponent >= 0 &&
+ mantissa <= binary_format<T>::max_mantissa_fast_path(exponent)) {
+#if defined(__clang__) || defined(SIMDJSON_FASTFLOAT_32BIT)
+ // Clang may map 0 to -0.0 when fegetround() == FE_DOWNWARD
+ if (mantissa == 0) {
+ value = is_negative ? T(-0.) : T(0.);
+ return true;
+ }
+#endif
+ value = T(mantissa) * binary_format<T>::exact_power_of_ten(exponent);
+ if (is_negative) {
+ value = -value;
+ }
+ return true;
+ }
}
- if (p - start_exp_digits == 0 || p - start_exp_digits > 19) {
- simdjson_consteval_error("Invalid float value");
+ }
+ return false;
+}
+
+/**
+ * This function overload takes parsed_number_string_t structure that is created
+ * and populated either by from_chars_advanced function taking chars range and
+ * parsing options or other parsing custom function implemented by user.
+ */
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(parsed_number_string_t<UC> &pns, T &value) noexcept {
+ static_assert(is_supported_float_type<T>::value,
+ "only some floating-point types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ from_chars_result_t<UC> answer;
+
+ answer.ec = std::errc(); // be optimistic
+ answer.ptr = pns.lastmatch;
+
+ if (!pns.too_many_digits &&
+ clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value))
+ return answer;
+
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+ if (pns.too_many_digits && am.power2 >= 0) {
+ if (am != compute_float<binary_format<T>>(pns.exponent, pns.mantissa + 1)) {
+ am = compute_error<binary_format<T>>(pns.exponent, pns.mantissa);
}
- exponent += exp_neg ? 0 - exp : exp;
}
+ // If we called compute_float<binary_format<T>>(pns.exponent, pns.mantissa)
+ // and we have an invalid power (am.power2 < 0), then we need to go the long
+ // way around again. This is very uncommon.
+ if (am.power2 < 0) {
+ am = digit_comp<T>(pns, am);
+ }
+ to_float(pns.negative, am, value);
+ // Test for over/underflow.
+ if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+ am.power2 == binary_format<T>::infinite_power()) {
+ answer.ec = std::errc::result_out_of_range;
+ }
+ return answer;
+}
- overflow = overflow || exponent < simdjson::internal::smallest_power ||
- exponent > simdjson::internal::largest_power;
- if (overflow) {
- simdjson_consteval_error("Overflow while computing the float value");
+// Slow path: re-parse materializing the integer/fraction spans the hot no-span
+// parse skipped, then run the full algorithm. The two callers reach it only
+// through a simdjson_fastfloat_unlikely branch, so the optimizer keeps this re-parse off
+// the hot path on its own (no function-level noinline needed).
+// from_chars_advanced already handles both the too_many_digits disambiguation
+// and the am.power2<0 digit_comp recompute, so both slow branches collapse to
+// one helper call.
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_number_slow_path(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options, bool bjf) noexcept {
+ parsed_number_string_t<UC> pns =
+ bjf ? parse_number_string<true, UC>(first, last, options, true)
+ : parse_number_string<false, UC>(first, last, options, true);
+ return from_chars_advanced(pns, value);
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_float_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+
+ static_assert(is_supported_float_type<T>::value,
+ "only some floating-point types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+
+ from_chars_result_t<UC> answer;
+ if (uint64_t(fmt & chars_format::skip_white_space)) {
+ while ((first != last) && simdjson_fast_float::is_space(*first)) {
+ first++;
+ }
+ }
+ if (first == last) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+ bool const bjf = uint64_t(fmt & detail::basic_json_fmt) != 0;
+
+ // Fast path: parse WITHOUT materializing the integer/fraction spans (read
+ // only by the rare slow paths). Skipping their stores keeps the fat
+ // parsed_number_string_t off the hot path. store_spans is a runtime argument,
+ // so this reuses the single parse_number_string instantiation.
+ parsed_number_string_t<UC> pns =
+ bjf ? parse_number_string<true, UC>(first, last, options, false)
+ : parse_number_string<false, UC>(first, last, options, false);
+ if (!pns.valid) {
+ if (uint64_t(fmt & chars_format::no_infnan)) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ } else {
+ return detail::parse_infnan(first, last, value, fmt);
+ }
}
- double d;
- if (!compute_float_64(exponent, i, negative, d)) {
+
+ // Slow path A (rare): > 19 significant digits. The no-span parse left the
+ // mantissa un-truncated and skipped the span-based recompute; the cold helper
+ // re-parses with spans and runs the full algorithm.
+ //
+// We have to disable -Wc++20-extensions for the [[unlikely]] attribute
+// See comment for @jwakely at
+// https://github.com/fastfloat/simdjson_fast_float/pull/387#discussion_r3366943539
+// This is unfortunate.
+#ifdef __clang__
+#pragma clang diagnostic push
+#if (!defined(__APPLE_CC__) && __clang_major__ >= 10) || (__clang_major__ >= 13)
+#pragma clang diagnostic ignored "-Wc++20-extensions"
+#endif
+#endif
+ if simdjson_fastfloat_unlikely (pns.too_many_digits) {
+ return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
+ }
+ answer.ec = std::errc(); // be optimistic
+ answer.ptr = pns.lastmatch;
+
+ if (clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value)) {
+ return answer;
+ }
+
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+ // Slow path B (rare): Eisel-Lemire could not resolve; digit_comp needs the
+ // integer/fraction spans. Route to the cold helper (clinger there is a
+ // dead-effect since it already failed here; the cold re-parse + digit_comp
+ // via from_chars_advanced reproduces this branch).
+ if simdjson_fastfloat_unlikely (am.power2 < 0) {
+ return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
+ }
+#ifdef __clang__
+#pragma clang diagnostic pop
+#endif
+ to_float(pns.negative, am, value);
+ // Test for over/underflow.
+ if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+ am.power2 == binary_format<T>::infinite_power()) {
+ answer.ec = std::errc::result_out_of_range;
+ }
+ return answer;
+}
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base) noexcept {
+
+ static_assert(is_supported_integer_type<T>::value,
+ "only integer types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ parse_options_t<UC> options;
+ options.base = base;
+ return from_chars_advanced(first, last, value, options);
+}
+
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+ T value;
+ if (clinger_fast_path_impl(mantissa, decimal_exponent, false, value))
+ return value;
+
+ adjusted_mantissa am =
+ compute_float<binary_format<T>>(decimal_exponent, mantissa);
+ to_float(false, am, value);
+ return value;
+}
+
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value, T>::type
+ integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+ const bool is_negative = mantissa < 0;
+ const uint64_t m = static_cast<uint64_t>(is_negative ? -mantissa : mantissa);
+
+ T value;
+ if (clinger_fast_path_impl(m, decimal_exponent, is_negative, value))
+ return value;
+
+ adjusted_mantissa am = compute_float<binary_format<T>>(decimal_exponent, m);
+ to_float(is_negative, am, value);
+ return value;
+}
+
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+// the following overloads are here to avoid surprising ambiguity for int,
+// unsigned, etc.
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value &&
+ std::is_integral<Int>::value &&
+ !std::is_signed<Int>::value,
+ T>::type
+ integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<T>(static_cast<uint64_t>(mantissa),
+ decimal_exponent);
+}
+
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+ typename std::enable_if<is_supported_float_type<T>::value &&
+ std::is_integral<Int>::value &&
+ std::is_signed<Int>::value,
+ T>::type
+ integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10<T>(static_cast<int64_t>(mantissa),
+ decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+ std::is_integral<Int>::value && !std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10(static_cast<uint64_t>(mantissa), decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+ std::is_integral<Int>::value && std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+ return integer_times_pow10(static_cast<int64_t>(mantissa), decimal_exponent);
+}
+
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_int_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+
+ static_assert(is_supported_integer_type<T>::value,
+ "only integer types are supported");
+ static_assert(is_supported_char_type<UC>::value,
+ "only char, wchar_t, char16_t and char32_t are supported");
+
+ chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+ int const base = options.base;
+
+ from_chars_result_t<UC> answer;
+ if (uint64_t(fmt & chars_format::skip_white_space)) {
+ while ((first != last) && simdjson_fast_float::is_space(*first)) {
+ first++;
+ }
+ }
+ if (first == last || base < 2 || base > 36) {
+ answer.ec = std::errc::invalid_argument;
+ answer.ptr = first;
+ return answer;
+ }
+
+ return parse_int_string(first, last, value, options);
+}
+
+template <size_t TypeIx> struct from_chars_advanced_caller {
+ static_assert(TypeIx > 0, "unsupported type");
+};
+
+template <> struct from_chars_advanced_caller<1> {
+ template <typename T, typename UC>
+ simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_float_advanced(first, last, value, options);
+ }
+};
+
+template <> struct from_chars_advanced_caller<2> {
+ template <typename T, typename UC>
+ simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+ call(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_int_advanced(first, last, value, options);
+ }
+};
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+ parse_options_t<UC> options) noexcept {
+ return from_chars_advanced_caller<
+ size_t(is_supported_float_type<T>::value) +
+ 2 * size_t(is_supported_integer_type<T>::value)>::call(first, last, value,
+ options);
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+/* end file simdjson/internal/fast_float.h */
+#include <array>
+#include <cstdint>
+#include <meta>
+#include <limits>
+#include <string_view>
+
+#include <algorithm>
+#include <array>
+#include <charconv>
+#include <cstdint>
+#include <expected>
+#include <meta>
+#include <string>
+#include <string_view>
+#include <vector>
+
+#define simdjson_consteval_error(...) \
+ { \
+ std::abort(); \
+ }
+
+namespace simdjson {
+namespace compile_time {
+
+/**
+ * Namespace for number parsing utilities.
+ * We seek to provide exact compile-time number parsing functions.
+ * Correct rounding of floating-point numbers is not a trivial matter, and it is
+ * harder still in a constant expression, where the runtime parser's memcpy and
+ * __uint128_t tricks are unavailable. We hand that part to fast_float, which is
+ * correctly rounded and usable in a constant expression from C++20 onwards.
+ */
+namespace number_parsing {
+
+// Parses a JSON float starting at src. Returns the value and the number of
+// characters consumed.
+//
+// Eisel-Lemire needs somewhere to fall back to: a mantissa of more than 19
+// significant digits, or an input where the truncated product cannot decide the
+// rounding, has to be finished by a slower exact method. A constant expression
+// cannot call the runtime one in src/from_chars.cpp, so we use fast_float, which
+// is correctly rounded in a constant expression and carries its own fallback.
+consteval std::pair<double, size_t> parse_double(const char *src,
+ const char *end) {
+ double value = 0;
+ auto answer = simdjson_fast_float::from_chars_advanced(
+ src, end, value,
+ simdjson_fast_float::parse_options{
+ simdjson_fast_float::chars_format::json});
+ if (answer.ec == std::errc::invalid_argument) {
+ simdjson_consteval_error("Invalid float value");
+ }
+ // fast_float reports result_out_of_range at either edge of the format. An
+ // underflow to zero is a value like any other; an overflow to infinity is not
+ // representable in JSON and was an error here before, so it stays one.
+ // (Comparing against max/lowest rather than calling std::isinf, which is not
+ // usable in a constant expression.)
+ if (value > (std::numeric_limits<double>::max)() ||
+ value < std::numeric_limits<double>::lowest()) {
simdjson_consteval_error("Overflow while computing the float value");
}
- return {d, size_t(p - srcinit)};
+ return {value, size_t(answer.ptr - src)};
}
+
} // namespace number_parsing
+consteval auto make_data_member_options(auto&& name_str) {
+ std::meta::data_member_options options{};
+ options.name = std::forward<decltype(name_str)>(name_str);
+ return options;
+}
+
// JSON string may contain embedded nulls, and C++26 reflection does not yet
// support std::string_view as a data member type. As a workaround, we define
// a custom type that holds a const char* and a size.
@@ -187056,7 +244847,8 @@ using class_type = type_builder<meta_info...>::constructed_type;
/**
* @brief Variable template for constructing instances with values
*/
-template <typename T, auto... Vs> constexpr auto construct_from = T{Vs...};
+template <typename T, auto... Vs> constexpr T construct_from = T{Vs...};
+
// in JSON, there are only a few whitespace characters that are allowed
// outside of objects, arrays, strings, and numbers.
@@ -187130,8 +244922,6 @@ parse_number(std::string_view json,
// Note that we consider -0 to be an integer unless it has a decimal point or
// exponent.
if (is_float) {
- // It would be cool to use std::from_chars in a consteval context, but it is
- // not supported yet for floating point types. :-(
auto [value, offset] =
number_parsing::parse_double(json.data(), json.data() + json.size());
if (offset != scope) {
@@ -187146,7 +244936,7 @@ parse_number(std::string_view json,
std::from_chars(json.data(), json.data() + json.size(), int_value);
if (res.ec == std::errc()) {
out = int_value;
- if ((res.ptr - json.data()) != scope) {
+ if (static_cast<std::size_t>(res.ptr - json.data()) != scope) {
simdjson_consteval_error(
"Internal error: cannot agree on the character range of the float");
}
@@ -187160,7 +244950,7 @@ parse_number(std::string_view json,
std::from_chars(json.data(), json.data() + json.size(), uint_value);
if (res.ec == std::errc()) {
out = uint_value;
- if ((res.ptr - json.data()) != scope) {
+ if (static_cast<std::size_t>(res.ptr - json.data()) != scope) {
simdjson_consteval_error(
"Internal error: cannot agree on the character range of the float");
}
@@ -187269,7 +245059,7 @@ parse_string(std::string_view json) {
// present, we have an error (isolated high surrogate), which we
// tolerate by substituting the substitution_code_point.
if (end - cursor < 6 || *cursor != '\\' ||
- *(cursor + 1) != 'u' > 0xFFFF) {
+ *(cursor + 1) != 'u') {
code_point = substitution_code_point;
} else { // we have \u following the high surrogate
cursor += 2; // skip \u
@@ -187623,10 +245413,19 @@ parse_json_array_impl(const std::string_view json) {
std::size_t count = values.size() - 1;
// We assume all elements have the same type as the first element.
// However, if the array is heterogeneous, we should use std::variant.
+ auto elem_type = std::meta::type_of(values[1]);
+ // String literals reflected via reflect_constant_string have type const
+ // char[N], but when passed as template auto parameters they decay to
+ // const char*. Use const char* as the element type so that
+ // construct_from can aggregate-initialize the array.
+ if (std::meta::is_array_type(elem_type) &&
+ std::meta::remove_all_extents(elem_type) == ^^const char) {
+ elem_type = ^^const char *;
+ }
auto array_type = std::meta::substitute(
^^std::array,
{
- std::meta::type_of(values[1]), std::meta::reflect_constant(count)});
+ elem_type, std::meta::reflect_constant(count)});
// Create array instance with values
values[0] = array_type;
@@ -187693,8 +245492,7 @@ parse_json_object_impl(std::string_view json) {
simdjson_consteval_error("Expected '}'");
}
cursor += object_size;
- auto dms = std::meta::data_member_spec(std::meta::type_of(parsed),
- {.name = field_name});
+ auto dms = std::meta::data_member_spec(std::meta::type_of(parsed), make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(parsed);
@@ -187703,8 +245501,7 @@ parse_json_object_impl(std::string_view json) {
case '[': {
std::string_view value(cursor, end);
auto [parsed, array_size] = parse_json_array_impl(value);
- auto dms = std::meta::data_member_spec(std::meta::type_of(parsed),
- {.name = field_name});
+ auto dms = std::meta::data_member_spec(std::meta::type_of(parsed), make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(parsed);
if (*(cursor + array_size - 1) != ']') {
@@ -187725,8 +245522,7 @@ parse_json_object_impl(std::string_view json) {
}
}
auto dms =
- std::meta::data_member_spec(^^const char *, {
- .name = field_name});
+ std::meta::data_member_spec(^^const char *, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant_string(value));
break;
@@ -187737,8 +245533,7 @@ parse_json_object_impl(std::string_view json) {
}
cursor += 4;
- auto dms = std::meta::data_member_spec(^^bool, {
- .name = field_name});
+ auto dms = std::meta::data_member_spec(^^bool, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(true));
break;
@@ -187749,8 +245544,7 @@ parse_json_object_impl(std::string_view json) {
}
cursor += 5;
- auto dms = std::meta::data_member_spec(^^bool, {
- .name = field_name});
+ auto dms = std::meta::data_member_spec(^^bool, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(false));
break;
@@ -187761,9 +245555,7 @@ parse_json_object_impl(std::string_view json) {
}
cursor += 4;
- auto dms = std::meta::data_member_spec(^^std::nullptr_t,
- {
- .name = field_name});
+ auto dms = std::meta::data_member_spec(^^std::nullptr_t, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(nullptr));
break;
@@ -187787,22 +245579,19 @@ parse_json_object_impl(std::string_view json) {
if (std::holds_alternative<int64_t>(out)) {
int64_t int_value = std::get<int64_t>(out);
auto dms =
- std::meta::data_member_spec(^^int64_t, {
- .name = field_name});
+ std::meta::data_member_spec(^^int64_t, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(int_value));
} else if (std::holds_alternative<uint64_t>(out)) {
uint64_t uint_value = std::get<uint64_t>(out);
auto dms =
- std::meta::data_member_spec(^^uint64_t, {
- .name = field_name});
+ std::meta::data_member_spec(^^uint64_t, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(uint_value));
} else {
double float_value = std::get<double>(out);
auto dms =
- std::meta::data_member_spec(^^double, {
- .name = field_name});
+ std::meta::data_member_spec(^^double, make_data_member_options(field_name));
members.push_back(std::meta::reflect_constant(dms));
values.push_back(std::meta::reflect_constant(float_value));
}
@@ -187847,16 +245636,11 @@ template <constevalutil::fixed_string json_str> consteval auto parse_json() {
"Only JSON objects and arrays are supported at the top level, this "
"limitation will be lifted in the future.");*/
- constexpr auto result = json.front() == '['
- ? parse_json_array_impl(json)
- : parse_json_object_impl(json);
- return [: result.first :];
- /*
- if(json.front() == '[') {
- return [:parse_json_array_impl(json).first:];
- } else if(json.front() == '{') {
- // return [:parse_json_object_impl(json).first:];
- }*/
+ if constexpr (json.front() == '[') {
+ return [: parse_json_array_impl(json).first :];
+ } else {
+ return [: parse_json_object_impl(json).first :];
+ }
}
} // namespace compile_time